unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,611 @@
|
|
|
1
|
+
"""Adapter for the ProntoQA dataset (Logic-LM's GPT-4 DSL rendering of Saparov
|
|
2
|
+
& He's PrOntoQA benchmark) -- local JSONL only, no network access.
|
|
3
|
+
|
|
4
|
+
Source and verified schema
|
|
5
|
+
---------------------------
|
|
6
|
+
Source: https://huggingface.co/datasets/renma/ProntoQA (unauthenticated;
|
|
7
|
+
verified directly via the Hugging Face ``datasets-server`` ``splits``/
|
|
8
|
+
``first-rows``/``rows`` APIs, 2026-08-12). Config ``"default"``, and -- unlike
|
|
9
|
+
FOLIO -- exactly ONE split, ``"validation"`` (500 rows; verified via
|
|
10
|
+
``datasets-server``'s ``info`` endpoint, which also confirms the file behind
|
|
11
|
+
it: ``hf://datasets/renma/ProntoQA/ProntoQA_dev_gpt-4.json``). There is no
|
|
12
|
+
train/test split to worry about defensively: this dataset has only the one.
|
|
13
|
+
|
|
14
|
+
That backing filename is not a coincidence: it is the SAME path
|
|
15
|
+
(``outputs/logic_programs/ProntoQA_dev_gpt-4.json``) published in the
|
|
16
|
+
Logic-LM paper's own repository, https://github.com/teacherpeterpan/Logic-LLM
|
|
17
|
+
(fetched 2026-08-12 to cross-check the DSL grammar below against the
|
|
18
|
+
REFERENCE solver, ``models/symbolic_solvers/pyke_solver/pyke_solver.py``).
|
|
19
|
+
So ``renma/ProntoQA`` is a re-upload of Logic-LM's own GPT-4-generated
|
|
20
|
+
symbolic-program outputs for the original PrOntoQA benchmark -- see
|
|
21
|
+
"Honesty: raw_logic_programs is LLM output, not ground truth" below for why
|
|
22
|
+
that provenance matters to this adapter's correctness guarantees.
|
|
23
|
+
|
|
24
|
+
Each verified row is a flat JSON object with exactly these keys (confirmed
|
|
25
|
+
against real ``rows``-API records):
|
|
26
|
+
|
|
27
|
+
* ``"id"`` -- ``str``, e.g. ``"ProntoQA_1"`` (always present
|
|
28
|
+
in every row fetched; this adapter still falls back positionally if a
|
|
29
|
+
record ever lacks it, mirroring :mod:`~unicode_logic_kit.eval.datasets.folio`).
|
|
30
|
+
* ``"answer"`` -- ``str``, always ``"A"`` or ``"B"`` in the
|
|
31
|
+
fetched sample, matching ``"options"`` (see below). NOT ``"True"``/
|
|
32
|
+
``"False"`` text -- kept verbatim as the source's own vocabulary, exactly
|
|
33
|
+
like FOLIO's ``"label"`` values are kept verbatim.
|
|
34
|
+
* ``"context"`` -- ``str``, the natural-language story (a single
|
|
35
|
+
paragraph, NOT pre-split into sentences -- there is no verified 1:1
|
|
36
|
+
mapping from an individual English sentence to a single DSL ``Rules:``
|
|
37
|
+
line, so this adapter does not attempt one; it maps the whole paragraph to
|
|
38
|
+
ONE ``nl_premises`` entry).
|
|
39
|
+
* ``"question"`` -- ``str``, e.g. ``"Is the following statement
|
|
40
|
+
true or false? Max is sour."``.
|
|
41
|
+
* ``"options"`` -- ``list[str]``, always exactly
|
|
42
|
+
``["A) True", "B) False"]`` in the fetched sample.
|
|
43
|
+
* ``"raw_logic_programs"`` -- ``Sequence[str]`` per the HF feature schema;
|
|
44
|
+
every row fetched while building this adapter carried EXACTLY one entry
|
|
45
|
+
(the schema's ``Sequence`` typing is presumably shared with a sibling
|
|
46
|
+
dataset that has more than one, not something ProntoQA itself exercises).
|
|
47
|
+
Each entry is Logic-LM's DSL rendering of the story, LLM-generated (GPT-4)
|
|
48
|
+
from the natural-language ``context`` -- see the grammar below, verified
|
|
49
|
+
against real rows AND against the reference Pyke solver's own parser
|
|
50
|
+
(``Pyke_Program.parse_logic_program``/``parse_forward_rule``/
|
|
51
|
+
``parse_query`` in ``pyke_solver.py``, cited above).
|
|
52
|
+
|
|
53
|
+
The Logic-LM DSL grammar (verified against real rows + the reference solver)
|
|
54
|
+
------------------------------------------------------------------------------
|
|
55
|
+
Four ``\\n\\n``-separated sections, in this fixed order (a fifth,
|
|
56
|
+
``"Predicates:"``, always comes first but this adapter IGNORES it entirely --
|
|
57
|
+
see below)::
|
|
58
|
+
|
|
59
|
+
Predicates:
|
|
60
|
+
Name($x, bool) ::: <english gloss>
|
|
61
|
+
...
|
|
62
|
+
|
|
63
|
+
Facts:
|
|
64
|
+
Predicate(Entity, True|False)
|
|
65
|
+
...
|
|
66
|
+
|
|
67
|
+
Rules:
|
|
68
|
+
Predicate($x, True|False) >>> Predicate($x, True|False)
|
|
69
|
+
...
|
|
70
|
+
|
|
71
|
+
Query:
|
|
72
|
+
Predicate(Entity, True|False)
|
|
73
|
+
|
|
74
|
+
* ``"Predicates:"`` lines just document a predicate's arity/gloss; VERIFIED
|
|
75
|
+
to sometimes be INCOMPLETE (``ProntoQA_4`` in the validation split lists
|
|
76
|
+
only 7 of the 17 predicate names its own ``Rules:``/``Facts:`` actually
|
|
77
|
+
use -- a real data-quality defect, not a transcription error in this
|
|
78
|
+
docstring). This adapter never reads this section: every predicate it
|
|
79
|
+
needs is discoverable directly from ``Facts:``/``Rules:``/``Query:``.
|
|
80
|
+
* ``"Facts:"`` -- one or more GROUND signed literals (every row fetched had
|
|
81
|
+
exactly one). ``Predicate(Entity, True)`` asserts the atom;
|
|
82
|
+
``Predicate(Entity, False)`` asserts its NEGATION -- the boolean is a
|
|
83
|
+
SIGN, not a third truth value (confirmed by cross-reading ``context``,
|
|
84
|
+
e.g. "Dumpuses are not wooden." renders as
|
|
85
|
+
``Dumpus($x, True) >>> Wooden($x, False)``).
|
|
86
|
+
* ``"Rules:"`` -- universally-quantified, single-variable Horn implications.
|
|
87
|
+
The reference solver's own ``parse_forward_rule`` splits both sides of
|
|
88
|
+
``>>>`` on ``&&`` for a MULTI-conjunct premise/conclusion (used by its
|
|
89
|
+
ProofWriter rules); ProntoQA rows verified here NEVER use ``&&`` --
|
|
90
|
+
always exactly one premise atom and one conclusion atom per rule -- but
|
|
91
|
+
this adapter still supports ``&&`` generically since it costs nothing and
|
|
92
|
+
keeps faith with the shared DSL grammar. ``$x`` is the only variable
|
|
93
|
+
placeholder ProntoQA rows were ever verified to use.
|
|
94
|
+
* ``"Query:"`` -- exactly one line, same ground-literal grammar as
|
|
95
|
+
``"Facts:"``.
|
|
96
|
+
|
|
97
|
+
Honesty: ``raw_logic_programs`` is LLM output, not ground truth
|
|
98
|
+
------------------------------------------------------------------
|
|
99
|
+
Because ``raw_logic_programs`` is a GPT-4 TRANSLATION of ``context`` (not a
|
|
100
|
+
hand-verified formalisation -- see the source discussion above), it can be,
|
|
101
|
+
and demonstrably sometimes is, internally defective in ways that are
|
|
102
|
+
INVISIBLE from ``answer`` alone. This was checked empirically: forward-
|
|
103
|
+
chaining ``Facts:``+``Rules:`` and comparing the derived value for the
|
|
104
|
+
``Query:`` predicate/subject against the ``Query:``'s own stated boolean --
|
|
105
|
+
exactly what the reference solver's ``answer_map_prontoqa`` computes (see
|
|
106
|
+
:func:`solve_example`) -- reproduces the gold ``answer`` on only 76 of a
|
|
107
|
+
100-row sample fetched from ``validation`` (row indices 0-99). The 24
|
|
108
|
+
mismatches split into two concretely identified classes:
|
|
109
|
+
|
|
110
|
+
1. A genuinely BROKEN derivation chain caused by an inconsistent predicate
|
|
111
|
+
name inside ``raw_logic_programs`` itself. E.g. ``ProntoQA_19``'s
|
|
112
|
+
``Rules:`` mix ``"Wumpus"``/``"Wumpuses"`` and
|
|
113
|
+
``"Tumpus"``/``"Tumpuses"``/``"Zumpus"``/``"Zumpuses"`` -- different
|
|
114
|
+
strings, so the chain the story clearly intends never actually connects,
|
|
115
|
+
and forward-chaining (both this adapter's, via :func:`solve_example`, and
|
|
116
|
+
presumably the reference Pyke solver's) derives NOTHING for the ``Query``
|
|
117
|
+
predicate.
|
|
118
|
+
2. Rows where the chain IS clean (no naming defect, single unambiguous
|
|
119
|
+
derivation) yet the derived value still does not match ``answer`` under
|
|
120
|
+
the reference formula. E.g. ``ProntoQA_1``: ``Query: Sour(Max, False)``;
|
|
121
|
+
forward-chaining ``Facts:``+``Rules:`` cleanly derives ``Sour(Max,
|
|
122
|
+
False)`` too (via ``Tumpus($x, True) >>> Sour($x, False)``, and
|
|
123
|
+
``Tumpus(Max, True)`` is itself the unique derived value along the
|
|
124
|
+
chain from the sole fact ``Yumpus(Max, True)``) -- so the reference
|
|
125
|
+
formula predicts ``'A'``, but the gold ``answer`` is ``'B'``. This
|
|
126
|
+
adapter does not know why (the original PrOntoQA generator and the exact
|
|
127
|
+
Logic-LM GPT-4 prompting/parsing script were not inspected further to
|
|
128
|
+
find out) and does NOT paper over it: :func:`solve_example` computes the
|
|
129
|
+
reference formula honestly and can legitimately disagree with ``answer``
|
|
130
|
+
on rows like this one.
|
|
131
|
+
|
|
132
|
+
Practical upshot: :func:`solve_example` is faithful to what
|
|
133
|
+
``raw_logic_programs`` literally entails -- it is NOT a guaranteed
|
|
134
|
+
reproducer of every gold ``answer`` in the full validation split. The
|
|
135
|
+
fixture shipped with this adapter (``tests/fixtures/prontoqa_mini.jsonl``)
|
|
136
|
+
was deliberately curated to 8 REAL rows (ids ProntoQA_4/9/18/22/31/39/45/52)
|
|
137
|
+
free of both defect classes above, so :func:`solve_example` DOES reproduce
|
|
138
|
+
``answer`` on every fixture row (see its own docstring for one fully
|
|
139
|
+
worked by hand).
|
|
140
|
+
|
|
141
|
+
This dataset does NOT have (being explicit, per this subpackage's honesty rule)
|
|
142
|
+
-------------------------------------------------------------------------------
|
|
143
|
+
* No FOL annotation at all: nothing in ``raw_logic_programs`` is a string
|
|
144
|
+
this kit's ``parse_any`` can parse (it is Logic-LM's own bracket DSL, not
|
|
145
|
+
this kit's unicode/TPTP/Prover9 surface syntax). Accordingly
|
|
146
|
+
``fol_premises`` is ALWAYS ``()`` and ``fol_conclusion`` is ALWAYS
|
|
147
|
+
``None`` on every :class:`~unicode_logic_kit.eval.datasets.DatasetExample`
|
|
148
|
+
this adapter yields -- the DSL text is preserved verbatim, unparsed, in
|
|
149
|
+
``meta["raw_logic_programs"]`` instead. A consequence:
|
|
150
|
+
:func:`~unicode_logic_kit.eval.datasets.audit_examples` run over
|
|
151
|
+
:func:`load_prontoqa`'s output is VACUOUSLY ``ok=True`` for every example
|
|
152
|
+
(nothing in ``fol_premises``/``fol_conclusion`` to fail parsing or
|
|
153
|
+
``check()``) -- it finds no defects here by construction, not because the
|
|
154
|
+
data has none (see the section above for where ProntoQA's real defects
|
|
155
|
+
actually live: inside the LLM-generated DSL, which ``audit_examples``
|
|
156
|
+
never looks at).
|
|
157
|
+
* No multi-hop branching or multi-entity relations: every ``Rules:`` line
|
|
158
|
+
is unary (single ``$x``), every predicate application has exactly one
|
|
159
|
+
argument, and every row verified here reasons about exactly one entity.
|
|
160
|
+
* Closed-world / deductively-complete WHEN THE DSL IS CLEAN: for a row
|
|
161
|
+
without the naming defect described above, forward-chaining derives
|
|
162
|
+
EXACTLY one signed literal for the ``Query``'s (predicate, subject) pair
|
|
163
|
+
-- either the ``Query`` itself or its negation, never neither and never
|
|
164
|
+
both. This is what lets :func:`solve_example` treat "not provable" and
|
|
165
|
+
"negation provable" as the two exhaustive, mutually exclusive outcomes
|
|
166
|
+
for a well-formed row (see its docstring) -- it is NOT a property this
|
|
167
|
+
adapter can guarantee for an arbitrary row, only something verified to
|
|
168
|
+
hold for the fixture.
|
|
169
|
+
|
|
170
|
+
License: **MIT**, per the Hugging Face repo's ``cardData.license`` field and
|
|
171
|
+
its ``license:mit`` tag (https://huggingface.co/api/datasets/renma/ProntoQA,
|
|
172
|
+
verified directly 2026-08-12). This loader only reads a LOCAL file the
|
|
173
|
+
caller already obtained; it never downloads or redistributes ProntoQA data.
|
|
174
|
+
|
|
175
|
+
This module never downloads anything -- obtain the JSONL file yourself from
|
|
176
|
+
the source above and pass its local path to :func:`load_prontoqa`.
|
|
177
|
+
"""
|
|
178
|
+
|
|
179
|
+
import re
|
|
180
|
+
from pathlib import Path
|
|
181
|
+
from typing import FrozenSet, Iterator, List, Sequence, Tuple, Union
|
|
182
|
+
|
|
183
|
+
from ._base import DatasetExample, _register_dataset_info
|
|
184
|
+
from ...fol.nodes import Node, Atom, Not, And, Implies, Quantifier, Variable, Constant
|
|
185
|
+
from ...atp.protocol import PROVED
|
|
186
|
+
|
|
187
|
+
__all__ = ["load_prontoqa", "parse_logic_program", "solve_example"]
|
|
188
|
+
|
|
189
|
+
_register_dataset_info(
|
|
190
|
+
"prontoqa",
|
|
191
|
+
license="MIT",
|
|
192
|
+
source_url="https://huggingface.co/datasets/renma/ProntoQA",
|
|
193
|
+
citation_hint=(
|
|
194
|
+
"Saparov, Abulhair, and He He. \"Language Models Are Greedy Reasoners: "
|
|
195
|
+
"A Systematic Formal Analysis of Chain-of-Thought.\" arXiv:2210.01240 "
|
|
196
|
+
"(the original PrOntoQA benchmark). DSL renderings (raw_logic_programs) "
|
|
197
|
+
"from Pan, Liangming, et al. \"Logic-LM: Empowering Large Language "
|
|
198
|
+
"Models with Symbolic Solvers for Faithful Logical Reasoning.\" "
|
|
199
|
+
"Findings of EMNLP 2023, arXiv:2305.12295 "
|
|
200
|
+
"(https://github.com/teacherpeterpan/Logic-LLM)."
|
|
201
|
+
),
|
|
202
|
+
)
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
# ---------------------------------------------------------------------------
|
|
206
|
+
# load_prontoqa -- JSONL -> DatasetExample
|
|
207
|
+
# ---------------------------------------------------------------------------
|
|
208
|
+
|
|
209
|
+
def _resolve_id(record: dict, line_no: int) -> str:
|
|
210
|
+
"""The record's own ``"id"`` if present, else a positional fallback.
|
|
211
|
+
|
|
212
|
+
Every row verified while building this adapter carried a native ``"id"``
|
|
213
|
+
(e.g. ``"ProntoQA_1"``); the fallback mirrors
|
|
214
|
+
:func:`~unicode_logic_kit.eval.datasets.folio._resolve_id` for a record
|
|
215
|
+
that, contrary to everything observed, lacks one.
|
|
216
|
+
"""
|
|
217
|
+
value = record.get("id")
|
|
218
|
+
if value is not None:
|
|
219
|
+
return str(value)
|
|
220
|
+
return f"prontoqa:{line_no}"
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def _example_from_record(record: dict, line_no: int,
|
|
224
|
+
known_bad_ids: FrozenSet[str]) -> DatasetExample:
|
|
225
|
+
context = record.get("context")
|
|
226
|
+
question = record.get("question")
|
|
227
|
+
answer = record.get("answer")
|
|
228
|
+
example_id = _resolve_id(record, line_no)
|
|
229
|
+
|
|
230
|
+
meta = {k: v for k, v in record.items()
|
|
231
|
+
if k not in ("id", "context", "question", "answer")}
|
|
232
|
+
if meta.get("options") is not None:
|
|
233
|
+
meta["options"] = tuple(meta["options"])
|
|
234
|
+
if meta.get("raw_logic_programs") is not None:
|
|
235
|
+
meta["raw_logic_programs"] = tuple(meta["raw_logic_programs"])
|
|
236
|
+
meta["line_no"] = line_no
|
|
237
|
+
|
|
238
|
+
return DatasetExample(
|
|
239
|
+
id=example_id,
|
|
240
|
+
nl_premises=(context,) if context is not None else (),
|
|
241
|
+
fol_premises=(), # see module docstring: DSL != FOL text
|
|
242
|
+
nl_conclusion=question,
|
|
243
|
+
fol_conclusion=None, # see module docstring: DSL != FOL text
|
|
244
|
+
label=answer, # kept verbatim ("A"/"B"), not translated
|
|
245
|
+
known_bad=example_id in known_bad_ids,
|
|
246
|
+
meta=meta,
|
|
247
|
+
)
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def load_prontoqa(path: Union[str, Path], *,
|
|
251
|
+
known_bad_ids: FrozenSet[str] = frozenset()) -> Iterator[DatasetExample]:
|
|
252
|
+
"""Stream :class:`~unicode_logic_kit.eval.datasets.DatasetExample` from a
|
|
253
|
+
local ProntoQA (renma/ProntoQA, ``validation`` split) JSONL file.
|
|
254
|
+
|
|
255
|
+
Args:
|
|
256
|
+
path: path to a local ``.jsonl`` file in the verified schema (see
|
|
257
|
+
module docstring) -- one JSON object per non-blank line. NEVER
|
|
258
|
+
downloaded by this function; obtain the file from
|
|
259
|
+
https://huggingface.co/datasets/renma/ProntoQA yourself.
|
|
260
|
+
known_bad_ids: ids (the record's own ``"id"``, e.g. ``"ProntoQA_1"``,
|
|
261
|
+
or the positional fallback ``f"prontoqa:{line_no}"``) whose
|
|
262
|
+
``raw_logic_programs``/``answer`` pairing is known to be
|
|
263
|
+
unreliable (e.g. an :func:`~unicode_logic_kit.eval.datasets.prontoqa.solve_example`
|
|
264
|
+
mismatch a caller has reviewed and wants flagged). Every yielded
|
|
265
|
+
example with a matching id gets ``known_bad=True``. Defaults to
|
|
266
|
+
an empty set.
|
|
267
|
+
|
|
268
|
+
Yields:
|
|
269
|
+
One :class:`~unicode_logic_kit.eval.datasets.DatasetExample` per
|
|
270
|
+
non-blank JSONL line, in file order. ``fol_premises`` is always
|
|
271
|
+
``()`` and ``fol_conclusion`` is always ``None`` (see module
|
|
272
|
+
docstring); the DSL text lives, verbatim and unparsed, in
|
|
273
|
+
``meta["raw_logic_programs"]``.
|
|
274
|
+
|
|
275
|
+
Raises:
|
|
276
|
+
FileNotFoundError: ``path`` does not exist.
|
|
277
|
+
json.JSONDecodeError: a non-blank line is not valid JSON -- this is
|
|
278
|
+
NOT swallowed; a malformed dataset file is a loud failure, not a
|
|
279
|
+
silently-skipped row (matches
|
|
280
|
+
:func:`~unicode_logic_kit.eval.datasets.folio.load_folio` /
|
|
281
|
+
:func:`~unicode_logic_kit.eval.datasets.malls.load_malls`).
|
|
282
|
+
"""
|
|
283
|
+
import json # local: mirrors folio.py/malls.py's own style
|
|
284
|
+
|
|
285
|
+
path = Path(path)
|
|
286
|
+
with path.open("r", encoding="utf-8") as fh:
|
|
287
|
+
for line_no, raw_line in enumerate(fh):
|
|
288
|
+
line = raw_line.strip()
|
|
289
|
+
if not line:
|
|
290
|
+
continue
|
|
291
|
+
record = json.loads(line)
|
|
292
|
+
yield _example_from_record(record, line_no, known_bad_ids)
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
# ---------------------------------------------------------------------------
|
|
296
|
+
# parse_logic_program -- the DSL -> kit-AST converter
|
|
297
|
+
# ---------------------------------------------------------------------------
|
|
298
|
+
|
|
299
|
+
_FORALL = "∀" # matches unicode_logic_kit.fol.modal_translation._FORALL
|
|
300
|
+
|
|
301
|
+
# A DSL atom: PredicateName(arg, True|False) -- arg is either a bare entity
|
|
302
|
+
# name (Facts:/Query:) or a "$x"-style variable placeholder (Rules:).
|
|
303
|
+
_ATOM_RE = re.compile(
|
|
304
|
+
r'^(?P<pred>[A-Za-z_][A-Za-z0-9_]*)\(\s*(?P<arg>[^,()]+?)\s*,\s*(?P<val>True|False)\s*\)$'
|
|
305
|
+
)
|
|
306
|
+
|
|
307
|
+
# The kit's own grammar (unicode_logic_kit/fol/grammars/terminals.lark), used here only
|
|
308
|
+
# to decide whether a DSL name is ALREADY a legal kit token or needs the documented
|
|
309
|
+
# fallback: PREDICATE = [A-Z][a-zA-Z0-9]*; a bare NAME constant = lowercase, >=2 letters.
|
|
310
|
+
_LEGAL_PREDICATE_RE = re.compile(r'^[A-Z][a-zA-Z0-9]*$')
|
|
311
|
+
_LEGAL_CONSTANT_RE = re.compile(r'^[a-z][a-zA-Z0-9]*[a-zA-Z][a-zA-Z0-9]*$')
|
|
312
|
+
|
|
313
|
+
_SECTION_MARKERS = ("Facts:", "Rules:", "Query:")
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
def _alnum(s: str) -> str:
|
|
317
|
+
return "".join(ch for ch in s if ch.isalnum())
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
def _sanitize_predicate(name: str) -> str:
|
|
321
|
+
"""DSL predicate names are already legal kit PREDICATE tokens
|
|
322
|
+
(``[A-Z][a-zA-Z0-9]*``) in every row this adapter was verified against --
|
|
323
|
+
this is a defensive fallback (force-capitalise, strip non-alnum) for
|
|
324
|
+
anything that is not, never exercised by the shipped fixture."""
|
|
325
|
+
if _LEGAL_PREDICATE_RE.match(name):
|
|
326
|
+
return name
|
|
327
|
+
base = _alnum(name) or "P"
|
|
328
|
+
if not base[0].isalpha():
|
|
329
|
+
base = "P" + base
|
|
330
|
+
return base[0].upper() + base[1:]
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
def _sanitize_constant(name: str) -> str:
|
|
334
|
+
"""Lowercase the DSL's bare entity name into the kit's plain multi-letter
|
|
335
|
+
NAME constant token: ``'Max' -> 'max'`` -- the convention this adapter
|
|
336
|
+
follows throughout (see module docstring), verified to succeed for every
|
|
337
|
+
entity name in the rows this adapter was built against (Max, Stella,
|
|
338
|
+
Wren, Alex, Fae, Sam, Sally, Rex, Polly, ...). Falls back to the kit's
|
|
339
|
+
explicit ``c_``-prefixed CONSTANT form for anything that would not
|
|
340
|
+
lowercase into a legal >=2-letter NAME (e.g. a single-letter entity
|
|
341
|
+
name) -- never exercised by the shipped fixture."""
|
|
342
|
+
lowered = name[:1].lower() + name[1:] if name else name
|
|
343
|
+
if _LEGAL_CONSTANT_RE.match(lowered):
|
|
344
|
+
return lowered
|
|
345
|
+
return "c_" + (_alnum(name) or "0")
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
def _nonblank_lines(text: str) -> List[str]:
|
|
349
|
+
return [line.strip() for line in text.splitlines() if line.strip()]
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
def _split_sections(program: str) -> Tuple[str, str, str]:
|
|
353
|
+
"""Split ``program`` into (Facts text, Rules text, Query text).
|
|
354
|
+
|
|
355
|
+
The leading ``"Predicates:"`` section (and anything else before
|
|
356
|
+
``"Facts:"``) is discarded -- see module docstring for why this adapter
|
|
357
|
+
never reads it. Raises a clear ``ValueError`` naming the first missing
|
|
358
|
+
marker rather than an opaque ``ValueError: not enough values to unpack``.
|
|
359
|
+
"""
|
|
360
|
+
remainder = program
|
|
361
|
+
sections: List[str] = []
|
|
362
|
+
for marker in _SECTION_MARKERS:
|
|
363
|
+
pieces = remainder.split(marker, 1)
|
|
364
|
+
if len(pieces) != 2:
|
|
365
|
+
raise ValueError(
|
|
366
|
+
f"parse_logic_program: missing {marker!r} section marker "
|
|
367
|
+
f"(expected 'Predicates:'/'Facts:'/'Rules:'/'Query:' sections "
|
|
368
|
+
f"in that order); program started with: {program[:120]!r}")
|
|
369
|
+
before, remainder = pieces
|
|
370
|
+
sections.append(before)
|
|
371
|
+
# sections[0] is the (discarded) Predicates section; sections[1]/[2] are
|
|
372
|
+
# the Facts/Rules text; `remainder` is everything after 'Query:'.
|
|
373
|
+
return sections[1], sections[2], remainder
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
def _parse_ground_literal(line: str) -> Tuple[Node, bool]:
|
|
377
|
+
"""Parse one ``Facts:``/``Query:`` line: ``Predicate(Entity, True|False)``.
|
|
378
|
+
|
|
379
|
+
Returns ``(node, polarity)`` where ``node`` is ``Atom(...)`` (polarity
|
|
380
|
+
True) or ``Not(Atom(...))`` (polarity False) and ``polarity`` is the raw
|
|
381
|
+
parsed boolean, returned separately for callers that want it without
|
|
382
|
+
re-inspecting the node (see :func:`parse_logic_program`).
|
|
383
|
+
"""
|
|
384
|
+
m = _ATOM_RE.match(line)
|
|
385
|
+
if not m:
|
|
386
|
+
raise ValueError(f"parse_logic_program: cannot parse fact/query line: {line!r}")
|
|
387
|
+
arg = m.group("arg").strip()
|
|
388
|
+
if arg.startswith("$"):
|
|
389
|
+
raise ValueError(
|
|
390
|
+
f"parse_logic_program: expected a ground constant in a Facts:/"
|
|
391
|
+
f"Query: line, found a Rules:-style variable placeholder {arg!r}: "
|
|
392
|
+
f"{line!r}")
|
|
393
|
+
predicate = _sanitize_predicate(m.group("pred"))
|
|
394
|
+
atom = Atom(predicate, (Constant(_sanitize_constant(arg)),))
|
|
395
|
+
polarity = m.group("val") == "True"
|
|
396
|
+
return (atom if polarity else Not(atom)), polarity
|
|
397
|
+
|
|
398
|
+
|
|
399
|
+
_RULE_VARIABLE = Variable("x")
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
def _parse_rule_literal(chunk: str) -> Node:
|
|
403
|
+
"""Parse one ``&&``-conjunct of a ``Rules:`` line's premise/conclusion:
|
|
404
|
+
``Predicate($x, True|False)``. Only ``$x`` is a verified placeholder for
|
|
405
|
+
ProntoQA (see module docstring); any other name raises rather than
|
|
406
|
+
silently mis-binding it to the wrong (or a fresh) variable."""
|
|
407
|
+
m = _ATOM_RE.match(chunk.strip())
|
|
408
|
+
if not m:
|
|
409
|
+
raise ValueError(f"parse_logic_program: cannot parse rule literal: {chunk!r}")
|
|
410
|
+
arg = m.group("arg").strip()
|
|
411
|
+
if arg != "$x":
|
|
412
|
+
raise ValueError(
|
|
413
|
+
f"parse_logic_program: unsupported rule variable {arg!r} -- every "
|
|
414
|
+
f"renma/ProntoQA rule verified while building this adapter used "
|
|
415
|
+
f"only '$x' as its single bound variable: {chunk!r}")
|
|
416
|
+
predicate = _sanitize_predicate(m.group("pred"))
|
|
417
|
+
atom = Atom(predicate, (_RULE_VARIABLE,))
|
|
418
|
+
return atom if m.group("val") == "True" else Not(atom)
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
def _conjoin(nodes: Sequence[Node]) -> Node:
|
|
422
|
+
result = nodes[0]
|
|
423
|
+
for n in nodes[1:]:
|
|
424
|
+
result = And(result, n)
|
|
425
|
+
return result
|
|
426
|
+
|
|
427
|
+
|
|
428
|
+
def _parse_rule(line: str) -> Node:
|
|
429
|
+
"""``Pred1($x, V1) [&& Pred2($x, V2) ...] >>> Pred3($x, V3) [&& ...]``
|
|
430
|
+
becomes ``∀x ( premise-conjunction(x) → conclusion-conjunction(x) )``.
|
|
431
|
+
``&&`` on either side is supported per the shared Logic-LM DSL grammar
|
|
432
|
+
(see :mod:`unicode_logic_kit.eval.datasets.prontoqa`'s module docstring)
|
|
433
|
+
though never observed in a verified ProntoQA row (only ever one premise,
|
|
434
|
+
one conclusion)."""
|
|
435
|
+
if ">>>" not in line:
|
|
436
|
+
raise ValueError(f"parse_logic_program: rule line has no '>>>': {line!r}")
|
|
437
|
+
premise_text, conclusion_text = line.split(">>>", 1)
|
|
438
|
+
premises = [_parse_rule_literal(c) for c in premise_text.split("&&")]
|
|
439
|
+
conclusions = [_parse_rule_literal(c) for c in conclusion_text.split("&&")]
|
|
440
|
+
body = Implies(_conjoin(premises), _conjoin(conclusions))
|
|
441
|
+
return Quantifier(_FORALL, _RULE_VARIABLE, body)
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
def parse_logic_program(program: str) -> Tuple[List[Node], Node, bool]:
|
|
445
|
+
"""Convert one ``raw_logic_programs`` DSL string into kit AST nodes.
|
|
446
|
+
|
|
447
|
+
Args:
|
|
448
|
+
program: one entry of ``example.meta["raw_logic_programs"]`` (the
|
|
449
|
+
full ``"Predicates:\\n...\\n\\nFacts:\\n...\\n\\nRules:\\n...\\n\\n
|
|
450
|
+
Query:\\n..."`` text -- see module docstring for the grammar).
|
|
451
|
+
|
|
452
|
+
Returns:
|
|
453
|
+
``(premises, query, query_polarity)`` where:
|
|
454
|
+
|
|
455
|
+
* ``premises`` is every ``Facts:`` ground literal followed by every
|
|
456
|
+
``Rules:`` implication, as kit :class:`~unicode_logic_kit.fol.nodes.Node`
|
|
457
|
+
objects, in source order.
|
|
458
|
+
* ``query`` is the ``Query:`` line's literal, AS STATED (its own
|
|
459
|
+
sign included) -- e.g. ``Sour(Max, False)`` becomes
|
|
460
|
+
``Not(Atom("Sour", (Constant("max"),)))``, not the bare positive
|
|
461
|
+
atom.
|
|
462
|
+
* ``query_polarity`` is the raw parsed boolean from the ``Query:``
|
|
463
|
+
line (``True`` for ``Sour(Max, True)``, ``False`` for
|
|
464
|
+
``Sour(Max, False)``), returned separately for convenience -- it
|
|
465
|
+
is exactly ``isinstance(query, Atom)`` restated as a bool, kept as
|
|
466
|
+
its own return value so a caller need not pattern-match the node.
|
|
467
|
+
|
|
468
|
+
Raises:
|
|
469
|
+
ValueError: a section marker is missing, a ``Facts:``/``Rules:``/
|
|
470
|
+
``Query:`` line does not match the DSL atom grammar, a
|
|
471
|
+
``Facts:``/``Query:`` line uses a ``$``-variable instead of a
|
|
472
|
+
ground constant, a ``Rules:`` line has no ``>>>``, a ``Rules:``
|
|
473
|
+
line uses a bound variable other than ``$x``, or the ``Query:``
|
|
474
|
+
section does not have exactly one line. Every case names the
|
|
475
|
+
offending text -- never a silent skip or a best-effort guess.
|
|
476
|
+
|
|
477
|
+
Naming convention (see module docstring for the full rationale):
|
|
478
|
+
predicates are used verbatim (already legal kit ``PREDICATE`` tokens in
|
|
479
|
+
every verified row); entity names are lowercased into the kit's bare
|
|
480
|
+
``NAME`` constant token (``"Max"`` -> ``"max"``); the DSL's ``$x`` becomes
|
|
481
|
+
the kit variable ``x``.
|
|
482
|
+
"""
|
|
483
|
+
facts_text, rules_text, query_text = _split_sections(program)
|
|
484
|
+
facts = [_parse_ground_literal(line)[0] for line in _nonblank_lines(facts_text)]
|
|
485
|
+
rules = [_parse_rule(line) for line in _nonblank_lines(rules_text)]
|
|
486
|
+
|
|
487
|
+
query_lines = _nonblank_lines(query_text)
|
|
488
|
+
if len(query_lines) != 1:
|
|
489
|
+
raise ValueError(
|
|
490
|
+
f"parse_logic_program: expected exactly one Query: line, got "
|
|
491
|
+
f"{len(query_lines)}: {query_lines!r}")
|
|
492
|
+
query, query_polarity = _parse_ground_literal(query_lines[0])
|
|
493
|
+
|
|
494
|
+
return facts + rules, query, query_polarity
|
|
495
|
+
|
|
496
|
+
|
|
497
|
+
# ---------------------------------------------------------------------------
|
|
498
|
+
# solve_example -- end to end via unicode_logic_kit.api.prove
|
|
499
|
+
# ---------------------------------------------------------------------------
|
|
500
|
+
|
|
501
|
+
def solve_example(example: DatasetExample, *, on_indefinite: str = "label",
|
|
502
|
+
**prove_kwargs) -> dict:
|
|
503
|
+
"""Decide one ProntoQA example end to end: DSL -> AST -> :func:`unicode_logic_kit.api.prove`.
|
|
504
|
+
|
|
505
|
+
Parses ``example.meta["raw_logic_programs"][0]`` via
|
|
506
|
+
:func:`parse_logic_program`, then asks
|
|
507
|
+
:func:`unicode_logic_kit.api.prove` whether ``premises`` entail the
|
|
508
|
+
``query`` literal EXACTLY as the DSL states it (sign included -- see
|
|
509
|
+
:func:`parse_logic_program`'s docstring). A row whose DSL is clean (see
|
|
510
|
+
module docstring's "does NOT have" section) is deductively closed: EITHER
|
|
511
|
+
the query or its negation is a classical consequence of ``premises``
|
|
512
|
+
(these are plain universally-quantified single-variable Horn
|
|
513
|
+
implications chained from one ground fact -- ordinary Modus Ponens,
|
|
514
|
+
nothing closed-world/non-classical about it), so ``PROVED``/``REFUTED``
|
|
515
|
+
are the two outcomes actually seen on the fixture. A row whose DSL has
|
|
516
|
+
the naming defect documented in the module docstring can legitimately
|
|
517
|
+
come back ``UNKNOWN`` instead (neither polarity is a consequence) --
|
|
518
|
+
this function does NOT guess in that case, it reports ``predicted=None``.
|
|
519
|
+
|
|
520
|
+
Worked example (``ProntoQA_45`` from the fixture, hand-verified):
|
|
521
|
+
``Facts: Dumpus(Fae, True)``. Chase the ``Rules:`` from that single
|
|
522
|
+
fact: ``Dumpus->Numpus->Zumpus->Wumpus->Impus``, and
|
|
523
|
+
``Impus($x, True) >>> Wooden($x, True)`` fires, giving the unique
|
|
524
|
+
derived literal ``Wooden(Fae, True)`` (the OTHER ``Wooden`` rule,
|
|
525
|
+
``Tumpus($x, True) >>> Wooden($x, False)``, never fires -- Fae is never
|
|
526
|
+
derived to be a Tumpus). ``Query: Wooden(Fae, True)`` is EXACTLY that
|
|
527
|
+
derived literal, so ``premises ⊨ query`` classically (a straight
|
|
528
|
+
Modus Ponens chain) -- ``api.prove`` returns ``PROVED``, this function
|
|
529
|
+
maps that to ``predicted="A"``, and the fixture's gold ``answer`` for
|
|
530
|
+
``ProntoQA_45`` is indeed ``"A"``.
|
|
531
|
+
|
|
532
|
+
Args:
|
|
533
|
+
example: a :class:`~unicode_logic_kit.eval.datasets.DatasetExample`
|
|
534
|
+
from :func:`load_prontoqa`.
|
|
535
|
+
**prove_kwargs: forwarded verbatim to
|
|
536
|
+
:func:`unicode_logic_kit.api.prove` (e.g. ``backends=[...]``,
|
|
537
|
+
``timeout=...``); with none given, ``prove`` uses its own
|
|
538
|
+
default chain, which is what the fixture was validated against.
|
|
539
|
+
|
|
540
|
+
Returns:
|
|
541
|
+
``{"predicted": "A" | "B" | None, "query_polarity": bool,``
|
|
542
|
+
``"verdict": Verdict.to_dict(), "verdict_negated": ... | None}``.
|
|
543
|
+
``predicted`` is ``"A"`` when ``premises ⊨ query`` is PROVED, ``"B"``
|
|
544
|
+
when ``premises ⊨ ¬query`` is PROVED (a SECOND prove call — a mere
|
|
545
|
+
REFUTED on the first call only witnesses that the entailment fails,
|
|
546
|
+
which is NOT a proof of the negation and is never coerced into
|
|
547
|
+
``"B"``), and ``None`` when neither direction closes
|
|
548
|
+
(``UNKNOWN``/``ERROR``/broken derivation chains -- never coerced
|
|
549
|
+
into a guess). ``verdict_negated`` carries the second call's verdict
|
|
550
|
+
dict, or ``None`` when the first call already settled it.
|
|
551
|
+
``query_polarity`` is :func:`parse_logic_program`'s third return
|
|
552
|
+
value, included so a caller can see the DSL's stated sign without
|
|
553
|
+
re-parsing.
|
|
554
|
+
|
|
555
|
+
``on_indefinite``: ProntoQA has only the two labels A/B, so a
|
|
556
|
+
neither-proved outcome is ALWAYS ``predicted=None`` — under
|
|
557
|
+
``"label"`` (default) and ``"abstain"`` alike (they coincide here,
|
|
558
|
+
unlike in the three-label ProverQA/ProofWriter solvers).
|
|
559
|
+
``"raise"`` additionally raises ``ValueError`` when a leg is
|
|
560
|
+
INDEFINITE (prover ``unknown``/``error`` — timeout, bound); it does
|
|
561
|
+
NOT raise when both directions are definitively REFUTED, since
|
|
562
|
+
"provably neither entailed" is an established (if unlabelable)
|
|
563
|
+
answer, not a prover failure.
|
|
564
|
+
|
|
565
|
+
Raises:
|
|
566
|
+
ValueError: ``example.meta["raw_logic_programs"]`` is missing, empty,
|
|
567
|
+
or has more than one entry (every row verified while building
|
|
568
|
+
this adapter carried exactly one -- see module docstring); or
|
|
569
|
+
:func:`parse_logic_program` itself raises on malformed DSL text.
|
|
570
|
+
"""
|
|
571
|
+
if on_indefinite not in ("label", "abstain", "raise"):
|
|
572
|
+
raise ValueError(
|
|
573
|
+
f"solve_example: on_indefinite must be 'label', 'abstain' or "
|
|
574
|
+
f"'raise', got {on_indefinite!r}")
|
|
575
|
+
programs = example.meta.get("raw_logic_programs") or ()
|
|
576
|
+
if len(programs) != 1:
|
|
577
|
+
raise ValueError(
|
|
578
|
+
f"solve_example: expected exactly one raw_logic_programs entry "
|
|
579
|
+
f"for {example.id!r}, got {len(programs)} (every renma/ProntoQA "
|
|
580
|
+
f"validation row inspected while building this adapter carried "
|
|
581
|
+
f"exactly one)")
|
|
582
|
+
|
|
583
|
+
premises, query, query_polarity = parse_logic_program(programs[0])
|
|
584
|
+
|
|
585
|
+
from ... import api # lazy: avoid import-time cost/cycles
|
|
586
|
+
|
|
587
|
+
verdict = api.prove(query, premises, **prove_kwargs)
|
|
588
|
+
if verdict.status == PROVED:
|
|
589
|
+
return {
|
|
590
|
+
"predicted": "A",
|
|
591
|
+
"query_polarity": query_polarity,
|
|
592
|
+
"verdict": verdict.to_dict(),
|
|
593
|
+
"verdict_negated": None,
|
|
594
|
+
}
|
|
595
|
+
# A REFUTED here only witnesses that premises ⊨ query FAILS (a
|
|
596
|
+
# countermodel to the entailment exists) — it is not a proof of the
|
|
597
|
+
# negation. "B" requires actually PROVING premises ⊨ ¬query.
|
|
598
|
+
negated = api.prove(Not(query), premises, **prove_kwargs)
|
|
599
|
+
if (negated.status != PROVED and on_indefinite == "raise"
|
|
600
|
+
and not (verdict.status == "refuted"
|
|
601
|
+
and negated.status == "refuted")):
|
|
602
|
+
raise ValueError(
|
|
603
|
+
f"solve_example: example {example.id}: indefinite prover outcome "
|
|
604
|
+
f"(query: {verdict.status}/{verdict.reason}, negated: "
|
|
605
|
+
f"{negated.status}/{negated.reason}) with on_indefinite='raise'.")
|
|
606
|
+
return {
|
|
607
|
+
"predicted": "B" if negated.status == PROVED else None,
|
|
608
|
+
"query_polarity": query_polarity,
|
|
609
|
+
"verdict": verdict.to_dict(),
|
|
610
|
+
"verdict_negated": negated.to_dict(),
|
|
611
|
+
}
|