unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,422 @@
|
|
|
1
|
+
"""Grammar and parser for ProofWriter's own proof-annotation strings.
|
|
2
|
+
|
|
3
|
+
ProofWriter (see :mod:`.proofwriter`'s module docstring for the dataset
|
|
4
|
+
itself) ships two DIFFERENT proof-annotation fields, both in the SAME
|
|
5
|
+
``triple``/``rule``-reference notation, kept verbatim and UNPARSED by every
|
|
6
|
+
loader in :mod:`.proofwriter`:
|
|
7
|
+
|
|
8
|
+
* ``question["proofs"]`` (structured/hitachi-nlp mirror, one entry per
|
|
9
|
+
question) — exactly ONE of two shapes:
|
|
10
|
+
|
|
11
|
+
(a) a positive/negative DERIVATION: a proof forest over the theory's own
|
|
12
|
+
named facts/rules. Leaves are bare ``tripleN`` references (a fact cited
|
|
13
|
+
directly); ``(t1 t2 …) -> ruleN`` applies ``ruleN`` to a CONJUNCTION of
|
|
14
|
+
antecedents, each ``ti`` itself either a bare ``tripleN`` leaf or a
|
|
15
|
+
further derivation (in which case it is written parenthesised, as an
|
|
16
|
+
``Item``); `` OR `` at the top level separates ALTERNATIVE derivations
|
|
17
|
+
of the same fact (a forest, not a single path). The WHOLE string is
|
|
18
|
+
wrapped in exactly one more pair of parens than its content strictly
|
|
19
|
+
needs (see :func:`parse_question_proof`'s docstring for the exact
|
|
20
|
+
grammar, hand-derived from the fixture below).
|
|
21
|
+
(b) for ``"Unknown"`` answers, a structurally DIFFERENT "deepest failure"
|
|
22
|
+
witness: ``@N: <fact text>.[CWA. Example of deepest failure =
|
|
23
|
+
(ruleK <- ruleJ <- … <- FAIL)]`` — a WALK-BACK chain of one or more
|
|
24
|
+
candidate rules, each tried and abandoned because ITS body's own
|
|
25
|
+
required atom was in turn undecidable (``ruleK`` is the rule whose
|
|
26
|
+
head could have produced ``<fact text>``; a chain of length 1 is the
|
|
27
|
+
common case, but real data goes up to at least 5 links deep — see
|
|
28
|
+
:class:`FailWitness`) — or the base case
|
|
29
|
+
``@N: <fact text>.[CWA. Example of deepest failure = (FAIL)]`` (no
|
|
30
|
+
candidate fact or rule head exists for ``<fact text>`` at all).
|
|
31
|
+
|
|
32
|
+
* ``"allProofs"`` (flat tasksource mirror, one string per THEORY, kept
|
|
33
|
+
verbatim in ``meta["allProofs"]`` by :func:`~.proofwriter.load_proofwriter`
|
|
34
|
+
and never otherwise touched) — a SEQUENCE of ``"@N: "``-headed sections
|
|
35
|
+
(one per proof-depth stratum), each listing every fact derivable at that
|
|
36
|
+
depth as ``"<fact text>.[<derivation>]"`` entries, space-separated. Every
|
|
37
|
+
``<derivation>`` observed here uses shape (a) above; shape (b) never
|
|
38
|
+
appears in ``allProofs`` (it only lists facts that WERE derived, not
|
|
39
|
+
questions that failed to derive).
|
|
40
|
+
|
|
41
|
+
Both fields are parsed by the SAME shape-(a) grammar
|
|
42
|
+
(:func:`_parse_derivation_string`); :func:`parse_question_proof` additionally
|
|
43
|
+
recognises shape (b), and :func:`parse_all_proofs` splits ``allProofs`` into
|
|
44
|
+
its per-depth sections before parsing each entry.
|
|
45
|
+
|
|
46
|
+
Grammar for shape (a), reverse-engineered here from the REAL fixture strings
|
|
47
|
+
in ``tests/fixtures/proofwriter_owa_mini.jsonl``/``proofwriter_cwa_mini.jsonl``
|
|
48
|
+
(every literal example in this docstring is copied verbatim from a fixture)
|
|
49
|
+
and confirmed against 390 additional real rows (5452 questions, every
|
|
50
|
+
published config) fetched from
|
|
51
|
+
``hitachi-nlp/proofwriter_processed_OWA`` (see ``check_gold_proof``'s test
|
|
52
|
+
coverage) — not guessed, worked out by matching parentheses one string at a
|
|
53
|
+
time::
|
|
54
|
+
|
|
55
|
+
TopTerm ::= "(" OrExpr ")" -- the whole annotation
|
|
56
|
+
OrExpr ::= Item ("OR" Item)* -- a forest of alternative derivations
|
|
57
|
+
Item ::= LEAF | "NAF" | "(" Deriv ")" -- LEAF/NAF bare, Deriv wrapped
|
|
58
|
+
Deriv ::= "(" Conj ")" "->" RULE -- a rule applied to a conjunction
|
|
59
|
+
Conj ::= Item+ -- one or more antecedents
|
|
60
|
+
LEAF ::= "triple" DIGIT+
|
|
61
|
+
RULE ::= "rule" DIGIT+
|
|
62
|
+
|
|
63
|
+
``NAF`` (a literal keyword, real CWA fixture
|
|
64
|
+
``tests/fixtures/proofwriter_cwa_mini.jsonl`` row ``RelNeg-CWA-D2-1420``
|
|
65
|
+
question ``Q4``: ``"[(((NAF) -> rule6))]"``) stands in for a ``tripleN``/
|
|
66
|
+
``ruleN`` reference at exactly one place: a rule's ``"~"``-polarity
|
|
67
|
+
(negation-as-failure) body condition the theory has no rule concluding the
|
|
68
|
+
negation of at all, so only the condition's plain ABSENCE justifies it, not
|
|
69
|
+
an explicit derivation — see :class:`Naf` and
|
|
70
|
+
:func:`.proofwriter.check_gold_proof`.
|
|
71
|
+
|
|
72
|
+
Worked examples (all four literally from the fixture, hand-parsed by
|
|
73
|
+
matching parens before this grammar was written down):
|
|
74
|
+
|
|
75
|
+
* ``"(triple6)"`` — TopTerm wraps a single bare LEAF: ``Leaf("triple6")``.
|
|
76
|
+
* ``"(((triple6) -> rule1))"`` — TopTerm wraps one Item, itself
|
|
77
|
+
``"(" Deriv ")"`` with ``Deriv = "(triple6) -> rule1"`` (Conj = one bare
|
|
78
|
+
LEAF): ``Apply("rule1", And((Leaf("triple6"),)))``.
|
|
79
|
+
* ``"(((((triple6) -> rule1) triple5) -> rule3))"`` — the outer Deriv's Conj
|
|
80
|
+
has TWO antecedents: the wrapped sub-derivation ``((triple6) -> rule1)``
|
|
81
|
+
and the bare leaf ``triple5``:
|
|
82
|
+
``Apply("rule3", And((Apply("rule1", And((Leaf("triple6"),))), Leaf("triple5"))))``.
|
|
83
|
+
* ``"(triple1 OR ((triple2) -> rule1))"`` — an OrExpr with two alternatives,
|
|
84
|
+
one bare LEAF and one wrapped Deriv:
|
|
85
|
+
``Or((Leaf("triple1"), Apply("rule1", And((Leaf("triple2"),)))))``.
|
|
86
|
+
|
|
87
|
+
An ``Or`` node is built only when there are two or more alternatives (a
|
|
88
|
+
single-alternative OrExpr collapses to that alternative directly, so the
|
|
89
|
+
tree never carries a spurious singleton ``Or``).
|
|
90
|
+
"""
|
|
91
|
+
|
|
92
|
+
import re
|
|
93
|
+
from dataclasses import dataclass
|
|
94
|
+
from typing import Dict, Tuple, Union
|
|
95
|
+
|
|
96
|
+
__all__ = [
|
|
97
|
+
"Leaf", "Naf", "And", "Apply", "Or", "FailWitness", "ProofNode",
|
|
98
|
+
"parse_question_proof", "parse_all_proofs",
|
|
99
|
+
]
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
# ---------------------------------------------------------------------------
|
|
103
|
+
# AST
|
|
104
|
+
# ---------------------------------------------------------------------------
|
|
105
|
+
|
|
106
|
+
@dataclass(frozen=True)
|
|
107
|
+
class Leaf:
|
|
108
|
+
"""A direct citation of one theory fact, e.g. ``Leaf("triple6")``."""
|
|
109
|
+
ref: str
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
@dataclass(frozen=True)
|
|
113
|
+
class Naf:
|
|
114
|
+
"""One antecedent satisfied by NEGATION AS FAILURE, with no explicit
|
|
115
|
+
fact/rule to cite: the literal ``"NAF"`` keyword real proofs use in
|
|
116
|
+
place of a ``tripleN``/``ruleN`` reference for a ``"~"``-polarity body
|
|
117
|
+
condition the theory has no rule concluding the negation of at all (so
|
|
118
|
+
only its plain ABSENCE, not an explicit derivation, justifies it — see
|
|
119
|
+
:func:`.proofwriter.check_gold_proof`'s docstring). Confirmed in the
|
|
120
|
+
real ``AllenAI`` CWA fixture (``tests/fixtures/proofwriter_cwa_mini.jsonl``,
|
|
121
|
+
row ``RelNeg-CWA-D2-1420``, question ``Q4``): ``"[(((NAF) -> rule6))]"``.
|
|
122
|
+
A no-field marker — there is nothing else to record.
|
|
123
|
+
"""
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
@dataclass(frozen=True)
|
|
127
|
+
class And:
|
|
128
|
+
"""The conjunction of antecedents a rule application is over.
|
|
129
|
+
|
|
130
|
+
``parts`` is non-empty and ORDER-preserving (index ``i`` is the ``i``-th
|
|
131
|
+
antecedent in the source string, matching the ``i``-th body literal of
|
|
132
|
+
the corresponding rule — see :func:`.proofwriter._as_rule`).
|
|
133
|
+
"""
|
|
134
|
+
parts: Tuple["ProofNode", ...]
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@dataclass(frozen=True)
|
|
138
|
+
class Apply:
|
|
139
|
+
"""One rule application: ``rule`` (e.g. ``"rule3"``) applied to ``args``."""
|
|
140
|
+
rule: str
|
|
141
|
+
args: And
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
@dataclass(frozen=True)
|
|
145
|
+
class Or:
|
|
146
|
+
"""A forest of two or more ALTERNATIVE derivations of the same fact."""
|
|
147
|
+
alts: Tuple["ProofNode", ...]
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
@dataclass(frozen=True)
|
|
151
|
+
class FailWitness:
|
|
152
|
+
"""The "deepest failure" witness ProofWriter records for an ``Unknown``
|
|
153
|
+
answer: neither ``atom`` nor its negation could be derived, and this is
|
|
154
|
+
ONE concrete reason why.
|
|
155
|
+
|
|
156
|
+
``rule_chain`` is the ordered walk-back of candidate rules the
|
|
157
|
+
generator's search tried and abandoned: ``()`` for the bare ``(FAIL)``
|
|
158
|
+
shape (no candidate fact or rule head existed for ``atom`` at all), or
|
|
159
|
+
one-or-more rule names (e.g. ``("rule6", "rule2")`` for
|
|
160
|
+
``"(rule6 <- rule2 <- FAIL)"``) — ``rule_chain[0]`` is the rule whose
|
|
161
|
+
HEAD could have produced ``atom``, ``rule_chain[1]`` is the rule that
|
|
162
|
+
could have produced ITS unsatisfied body atom, and so on down to a final
|
|
163
|
+
dead end. Confirmed against 300 real rows fetched from
|
|
164
|
+
``hitachi-nlp/proofwriter_processed_OWA`` (depth-0/1/2/3/5/NatLang) —
|
|
165
|
+
the single-rule shape the fixture alone suggested is the ``len == 1``
|
|
166
|
+
case of this more general chain, not the whole grammar.
|
|
167
|
+
``depth`` is the ``@N`` stratum the generator's search had reached.
|
|
168
|
+
"""
|
|
169
|
+
atom: str
|
|
170
|
+
rule_chain: Tuple[str, ...]
|
|
171
|
+
depth: int
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
ProofNode = Union[Leaf, Naf, Apply, Or, FailWitness]
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
# ---------------------------------------------------------------------------
|
|
178
|
+
# Shape (a): the derivation-term grammar
|
|
179
|
+
# ---------------------------------------------------------------------------
|
|
180
|
+
|
|
181
|
+
_LEAF_RE = re.compile(r"triple\d+\Z")
|
|
182
|
+
_RULE_RE = re.compile(r"rule\d+\Z")
|
|
183
|
+
_TERM_TOKEN_RE = re.compile(r"\(|\)|->|OR|[A-Za-z][A-Za-z0-9]*")
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def _tokenise_derivation(text: str) -> "list":
|
|
187
|
+
tokens = _TERM_TOKEN_RE.findall(text)
|
|
188
|
+
remainder = _TERM_TOKEN_RE.sub("", text).strip()
|
|
189
|
+
if remainder:
|
|
190
|
+
raise ValueError(
|
|
191
|
+
f"proofwriter: unrecognised material {remainder!r} in proof "
|
|
192
|
+
f"derivation {text!r} — outside the documented "
|
|
193
|
+
"TopTerm/OrExpr/Item/Deriv/Conj grammar")
|
|
194
|
+
return tokens
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _expect(tokens, pos, token, context):
|
|
198
|
+
if pos >= len(tokens) or tokens[pos] != token:
|
|
199
|
+
got = tokens[pos] if pos < len(tokens) else "<end>"
|
|
200
|
+
raise ValueError(
|
|
201
|
+
f"proofwriter: expected {token!r} {context}, got {got!r} in "
|
|
202
|
+
f"proof derivation tokens {tokens!r}")
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _parse_item(tokens, pos):
|
|
206
|
+
"""``Item ::= LEAF | "NAF" | "(" Deriv ")"``."""
|
|
207
|
+
if pos >= len(tokens):
|
|
208
|
+
raise ValueError(
|
|
209
|
+
f"proofwriter: proof derivation ends mid-item in {tokens!r}")
|
|
210
|
+
token = tokens[pos]
|
|
211
|
+
if token == "NAF":
|
|
212
|
+
return Naf(), pos + 1
|
|
213
|
+
if _LEAF_RE.match(token):
|
|
214
|
+
return Leaf(token), pos + 1
|
|
215
|
+
if token == "(":
|
|
216
|
+
# This '(' is the Item's OWN wrapper -- Deriv has its own separate
|
|
217
|
+
# leading '(' (for its Conj), consumed by _parse_deriv itself.
|
|
218
|
+
node, pos = _parse_deriv(tokens, pos + 1)
|
|
219
|
+
_expect(tokens, pos, ")", "to close an Item")
|
|
220
|
+
return node, pos + 1
|
|
221
|
+
if _RULE_RE.match(token):
|
|
222
|
+
raise ValueError(
|
|
223
|
+
f"proofwriter: rule reference {token!r} used where a fact/"
|
|
224
|
+
f"derivation Item was expected in {tokens!r} — outside the "
|
|
225
|
+
"documented grammar (a rule name only ever follows '->')")
|
|
226
|
+
raise ValueError(
|
|
227
|
+
f"proofwriter: {token!r} is not a well-formed Item (expected a "
|
|
228
|
+
f"'tripleN' leaf or a parenthesised derivation) in {tokens!r}")
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _parse_deriv(tokens, pos):
|
|
232
|
+
"""``Deriv ::= "(" Conj ")" "->" RULE``, ``Conj ::= Item+``."""
|
|
233
|
+
_expect(tokens, pos, "(", "to start a conjunction")
|
|
234
|
+
pos += 1
|
|
235
|
+
items = []
|
|
236
|
+
while pos < len(tokens) and tokens[pos] != ")":
|
|
237
|
+
item, pos = _parse_item(tokens, pos)
|
|
238
|
+
items.append(item)
|
|
239
|
+
if not items:
|
|
240
|
+
raise ValueError(
|
|
241
|
+
f"proofwriter: empty conjunction '()' in proof derivation "
|
|
242
|
+
f"{tokens!r} — a rule application needs at least one antecedent")
|
|
243
|
+
_expect(tokens, pos, ")", "to close a conjunction")
|
|
244
|
+
pos += 1
|
|
245
|
+
_expect(tokens, pos, "->", "after a conjunction")
|
|
246
|
+
pos += 1
|
|
247
|
+
if pos >= len(tokens) or not _RULE_RE.match(tokens[pos]):
|
|
248
|
+
got = tokens[pos] if pos < len(tokens) else "<end>"
|
|
249
|
+
raise ValueError(
|
|
250
|
+
f"proofwriter: expected a 'ruleN' reference after '->', got "
|
|
251
|
+
f"{got!r} in proof derivation {tokens!r}")
|
|
252
|
+
rule = tokens[pos]
|
|
253
|
+
return Apply(rule, And(tuple(items))), pos + 1
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def _parse_or_expr(tokens, pos):
|
|
257
|
+
"""``OrExpr ::= Item ("OR" Item)*``."""
|
|
258
|
+
alts = []
|
|
259
|
+
item, pos = _parse_item(tokens, pos)
|
|
260
|
+
alts.append(item)
|
|
261
|
+
while pos < len(tokens) and tokens[pos] == "OR":
|
|
262
|
+
pos += 1
|
|
263
|
+
item, pos = _parse_item(tokens, pos)
|
|
264
|
+
alts.append(item)
|
|
265
|
+
if len(alts) == 1:
|
|
266
|
+
return alts[0], pos
|
|
267
|
+
return Or(tuple(alts)), pos
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def _parse_derivation_string(text: str) -> "ProofNode":
|
|
271
|
+
"""``TopTerm ::= "(" OrExpr ")"`` over the WHOLE ``text``."""
|
|
272
|
+
tokens = _tokenise_derivation(text)
|
|
273
|
+
if not tokens:
|
|
274
|
+
raise ValueError("proofwriter: empty proof derivation")
|
|
275
|
+
_expect(tokens, 0, "(", "at the start of a proof derivation")
|
|
276
|
+
node, pos = _parse_or_expr(tokens, 1)
|
|
277
|
+
_expect(tokens, pos, ")", "to close the proof derivation")
|
|
278
|
+
pos += 1
|
|
279
|
+
if pos != len(tokens):
|
|
280
|
+
raise ValueError(
|
|
281
|
+
f"proofwriter: trailing tokens {tokens[pos:]!r} after a "
|
|
282
|
+
f"complete proof derivation in {text!r}")
|
|
283
|
+
return node
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
# ---------------------------------------------------------------------------
|
|
287
|
+
# Shape (b): the "deepest failure" witness
|
|
288
|
+
# ---------------------------------------------------------------------------
|
|
289
|
+
|
|
290
|
+
_FAIL_WITNESS_RE = re.compile(
|
|
291
|
+
r"\A@(?P<depth>\d+):\s*(?P<atom>.+?)\.\[CWA\. Example of deepest "
|
|
292
|
+
r"failure = \((?P<chain>(?:rule\d+ <- )*)FAIL\)\]\Z"
|
|
293
|
+
)
|
|
294
|
+
_CHAIN_RULE_RE = re.compile(r"rule\d+")
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def _parse_fail_witness(text: str) -> "FailWitness":
|
|
298
|
+
match = _FAIL_WITNESS_RE.match(text)
|
|
299
|
+
if not match:
|
|
300
|
+
raise ValueError(
|
|
301
|
+
f"proofwriter: {text!r} starts with '@N:' but does not match "
|
|
302
|
+
"the documented 'deepest failure' witness shape "
|
|
303
|
+
"'@N: <fact>.[CWA. Example of deepest failure = "
|
|
304
|
+
"(ruleK <- ruleJ <- … <- FAIL) | (FAIL)]'")
|
|
305
|
+
chain = tuple(_CHAIN_RULE_RE.findall(match.group("chain")))
|
|
306
|
+
return FailWitness(atom=match.group("atom"), rule_chain=chain,
|
|
307
|
+
depth=int(match.group("depth")))
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
# ---------------------------------------------------------------------------
|
|
311
|
+
# Public entry points
|
|
312
|
+
# ---------------------------------------------------------------------------
|
|
313
|
+
|
|
314
|
+
def parse_question_proof(text: str) -> "ProofNode":
|
|
315
|
+
"""One ``question["proofs"]`` string → a :data:`ProofNode`.
|
|
316
|
+
|
|
317
|
+
Accepts both documented shapes: a bracketed derivation term (shape (a),
|
|
318
|
+
→ :class:`Leaf` / :class:`Apply` / :class:`Or`) and a bracketed "deepest
|
|
319
|
+
failure" witness (shape (b), → :class:`FailWitness`). The distinguishing
|
|
320
|
+
mark is the FIRST character inside the brackets: ``"@"`` selects shape
|
|
321
|
+
(b), anything else shape (a) — exactly how the two shapes are told apart
|
|
322
|
+
in the real data (a derivation never starts with ``"@"``, a witness
|
|
323
|
+
always does).
|
|
324
|
+
|
|
325
|
+
Args:
|
|
326
|
+
text: the raw string, e.g. ``"[(triple6)]"`` or
|
|
327
|
+
``"[@0: Charlie is round.[CWA. Example of deepest failure = "
|
|
328
|
+
"(FAIL)]]"``.
|
|
329
|
+
|
|
330
|
+
Raises:
|
|
331
|
+
ValueError: ``text`` is not wrapped in exactly one pair of ``[]``,
|
|
332
|
+
its bracketed content matches neither documented shape, or it is
|
|
333
|
+
the bare ``"[]"`` some hand-authored ``birds-electricity``/
|
|
334
|
+
``NatLang`` theories use for an ``"Unknown"`` question with NO
|
|
335
|
+
recorded witness at all (confirmed against 30 real
|
|
336
|
+
``birds-electricity`` rows: 822 such cases, every one
|
|
337
|
+
``answer == "Unknown"`` — a real annotation shape, but one this
|
|
338
|
+
grammar does not cover, so it is named and refused rather than
|
|
339
|
+
silently treated as an empty derivation or a witness with no
|
|
340
|
+
content). Every error names the unrecognised material or shape,
|
|
341
|
+
never guesses.
|
|
342
|
+
"""
|
|
343
|
+
if not (text.startswith("[") and text.endswith("]") and len(text) >= 2):
|
|
344
|
+
raise ValueError(
|
|
345
|
+
f"proofwriter: proof annotation {text!r} is not wrapped in "
|
|
346
|
+
"'[...]' as every 'proofs' field value is in the fixtures "
|
|
347
|
+
"checked here")
|
|
348
|
+
inner = text[1:-1]
|
|
349
|
+
if inner == "":
|
|
350
|
+
raise ValueError(
|
|
351
|
+
"proofwriter: proof annotation '[]' carries no derivation and "
|
|
352
|
+
"no failure witness (seen in the hand-authored birds-electricity/"
|
|
353
|
+
"NatLang configs for some 'Unknown' questions) — outside the "
|
|
354
|
+
"documented derivation/failure-witness grammar; refusing rather "
|
|
355
|
+
"than guessing at a reason")
|
|
356
|
+
if inner.startswith("@"):
|
|
357
|
+
return _parse_fail_witness(inner)
|
|
358
|
+
return _parse_derivation_string(inner)
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
_SECTION_RE = re.compile(r"@(\d+):")
|
|
362
|
+
_ALLPROOFS_ENTRY_RE = re.compile(r"\s*(?P<text>.+?\.)\[(?P<term>[^\[\]]*)\]")
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
def parse_all_proofs(text: str) -> "Dict[int, Dict[str, ProofNode]]":
|
|
366
|
+
"""The theory-wide ``allProofs`` string → ``{depth: {fact_text: node}}``.
|
|
367
|
+
|
|
368
|
+
``allProofs`` (see :mod:`.proofwriter`'s module docstring — carried
|
|
369
|
+
verbatim in ``meta["allProofs"]`` by :func:`~.proofwriter.load_proofwriter`
|
|
370
|
+
and never otherwise parsed there) lists, for each proof-depth stratum
|
|
371
|
+
``@N``, every fact derivable at that depth as ``"<fact text>.[<term>]"``
|
|
372
|
+
entries; every ``<term>`` observed here is shape (a) (see module
|
|
373
|
+
docstring) — this function does not expect shape (b) inside ``allProofs``
|
|
374
|
+
and raises if one is found (``allProofs`` only records SUCCESSFUL
|
|
375
|
+
derivations, never failure witnesses).
|
|
376
|
+
|
|
377
|
+
This is a STANDALONE parse: ``load_proofwriter`` generates no FOL for
|
|
378
|
+
these theories (see its docstring's "Honesty" section), so there is
|
|
379
|
+
nothing to cross-check an ``allProofs`` derivation against — unlike
|
|
380
|
+
:func:`.proofwriter.check_gold_proof`, which verifies
|
|
381
|
+
``question["proofs"]`` from the structured route against the kit's own
|
|
382
|
+
forward-chaining fixpoint.
|
|
383
|
+
|
|
384
|
+
Raises:
|
|
385
|
+
ValueError: ``text`` contains material outside the
|
|
386
|
+
``"@N: fact.[term] fact.[term] …"`` shape, naming the
|
|
387
|
+
unrecognised fragment and its stratum.
|
|
388
|
+
"""
|
|
389
|
+
result: "Dict[int, Dict[str, ProofNode]]" = {}
|
|
390
|
+
sections = list(_SECTION_RE.finditer(text))
|
|
391
|
+
if not sections:
|
|
392
|
+
raise ValueError(
|
|
393
|
+
f"proofwriter: allProofs string has no '@N:' section header at "
|
|
394
|
+
f"all: {text!r}")
|
|
395
|
+
for index, marker in enumerate(sections):
|
|
396
|
+
depth = int(marker.group(1))
|
|
397
|
+
start = marker.end()
|
|
398
|
+
end = sections[index + 1].start() if index + 1 < len(sections) else len(text)
|
|
399
|
+
section = text[start:end]
|
|
400
|
+
entries: "Dict[str, ProofNode]" = {}
|
|
401
|
+
pos = 0
|
|
402
|
+
for entry in _ALLPROOFS_ENTRY_RE.finditer(section):
|
|
403
|
+
if entry.start() != pos:
|
|
404
|
+
raise ValueError(
|
|
405
|
+
f"proofwriter: unrecognised material "
|
|
406
|
+
f"{section[pos:entry.start()]!r} in allProofs @{depth} "
|
|
407
|
+
f"section {section!r}")
|
|
408
|
+
fact_text = entry.group("text")
|
|
409
|
+
term = entry.group("term")
|
|
410
|
+
if term.strip().startswith("@"):
|
|
411
|
+
raise ValueError(
|
|
412
|
+
f"proofwriter: allProofs entry for {fact_text!r} at "
|
|
413
|
+
f"@{depth} looks like a failure witness ({term!r}), "
|
|
414
|
+
"which allProofs is not documented to ever contain")
|
|
415
|
+
entries[fact_text] = _parse_derivation_string(term)
|
|
416
|
+
pos = entry.end()
|
|
417
|
+
if section[pos:].strip():
|
|
418
|
+
raise ValueError(
|
|
419
|
+
f"proofwriter: unrecognised trailing material "
|
|
420
|
+
f"{section[pos:]!r} in allProofs @{depth} section")
|
|
421
|
+
result[depth] = entries
|
|
422
|
+
return result
|