unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
"""Adapter for the GROVES dataset (Vossel, "Groves" dataset card,
|
|
2
|
+
https://huggingface.co/datasets/fvossel/groves) — local JSONL only, no
|
|
3
|
+
network access.
|
|
4
|
+
|
|
5
|
+
Source and verified schema
|
|
6
|
+
---------------------------
|
|
7
|
+
Source: https://huggingface.co/datasets/fvossel/groves (unauthenticated;
|
|
8
|
+
verified 2026-08-12 directly via the Hugging Face datasets-server
|
|
9
|
+
``/splits``, ``/first-rows``, and ``/size`` APIs, and cross-checked against
|
|
10
|
+
the raw upstream ``train.json`` file via ``huggingface.co/.../resolve/main/``).
|
|
11
|
+
Config ``default`` has three splits — ``train`` (35910 rows), ``validation``
|
|
12
|
+
(3989 rows), ``test`` (9971 rows) — each row a flat JSON object with exactly
|
|
13
|
+
two keys:
|
|
14
|
+
|
|
15
|
+
* ``"NL"`` — ``str``, one natural-language statement.
|
|
16
|
+
* ``"FOL"`` — ``str``, its FOL translation, already written in THIS KIT'S OWN
|
|
17
|
+
unicode surface syntax (``∀``/``∃``/``∧``/``∨``/``¬``/``→``/``↔``/``⊕``/
|
|
18
|
+
``≠``/``=`` all appear in the 100-row verified sample and ALL parse under
|
|
19
|
+
this kit's default ``fol`` mode via :func:`unicode_logic_kit.api.parse_any` —
|
|
20
|
+
confirmed directly, not assumed, by parsing and :func:`~unicode_logic_kit.api.check`-
|
|
21
|
+
ing a 26-row spot sample spanning every connective/quantifier combination
|
|
22
|
+
seen in the sample, all of which came back ``ok=True``/``is_closed=True``/
|
|
23
|
+
``arity_consistent=True``/``has_lambdas=False``). GROVES is the only
|
|
24
|
+
adapter in this subpackage whose FOL column needs no dialect coaxing at
|
|
25
|
+
all — it was generated directly in this kit's notation.
|
|
26
|
+
|
|
27
|
+
GROVES has NO id field, NO premises/conclusion structure, and NO entailment
|
|
28
|
+
label — the SAME flat (NL, FOL) translation-pair shape as
|
|
29
|
+
:mod:`~unicode_logic_kit.eval.datasets.malls`, mapped onto
|
|
30
|
+
:class:`~unicode_logic_kit.eval.datasets.DatasetExample` the same way (see
|
|
31
|
+
``malls.py``'s docstring for the field-mapping rationale, reused verbatim
|
|
32
|
+
here):
|
|
33
|
+
|
|
34
|
+
* ``nl_conclusion`` / ``fol_conclusion`` carry ``"NL"`` / ``"FOL"``.
|
|
35
|
+
* ``nl_premises`` / ``fol_premises`` are always ``()``.
|
|
36
|
+
* ``label`` is always ``None``.
|
|
37
|
+
|
|
38
|
+
Provenance (per the dataset card's "Description"/"Licensing" sections,
|
|
39
|
+
verified 2026-08-12): GROVES's natural-language inputs are drawn from
|
|
40
|
+
WillowNLtoFOL (https://huggingface.co/datasets/iedeveci/WillowNLtoFOL,
|
|
41
|
+
originally CC BY-NC-ND 4.0, "included with permission from the original
|
|
42
|
+
authors" per the card) and MALLS-v0
|
|
43
|
+
(https://huggingface.co/datasets/yuan-yang/MALLS-v0, CC BY-NC 4.0); the FOL
|
|
44
|
+
expressions themselves were newly generated for GROVES and the dataset was
|
|
45
|
+
"subsequently filtered" — the card gives no further detail on that
|
|
46
|
+
generation/filtering pipeline, and this adapter does not invent any (no
|
|
47
|
+
independent semantic re-verification of the FOL against the NL is claimed
|
|
48
|
+
here beyond "it parses and is well-formed under this kit").
|
|
49
|
+
|
|
50
|
+
Upstream GROVES is distributed as JSON ARRAY files (``train.json`` /
|
|
51
|
+
``val.json`` / ``test.json`` — one big ``[...]`` list per split, confirmed
|
|
52
|
+
directly from the raw file's first bytes), NOT as JSONL. Like
|
|
53
|
+
:mod:`~unicode_logic_kit.eval.datasets.malls`, this loader reads local JSONL
|
|
54
|
+
(one JSON object per line) for a uniform streaming adapter surface; convert
|
|
55
|
+
an upstream split file first (e.g. ``jq -c '.[]' train.json >
|
|
56
|
+
groves_train.jsonl``) before calling :func:`load_groves`.
|
|
57
|
+
|
|
58
|
+
License: **CC-BY-NC-4.0** (non-commercial), per the dataset card's ``license``
|
|
59
|
+
front-matter and its "Licensing" section — which additionally requires
|
|
60
|
+
downstream users to also comply with the licenses of the two datasets GROVES
|
|
61
|
+
is built from (WillowNLtoFOL, CC BY-NC-ND 4.0; MALLS-v0, CC BY-NC 4.0). This
|
|
62
|
+
loader itself has no license implications beyond reading a local file the
|
|
63
|
+
caller already obtained; it never downloads or redistributes GROVES data.
|
|
64
|
+
|
|
65
|
+
Citation: as of 2026-08-12 the dataset card's "Notes" section states only
|
|
66
|
+
that "[f]urther details about dataset construction and evaluation will be
|
|
67
|
+
provided in a forthcoming publication" — it does NOT name a paper. A
|
|
68
|
+
plausible companion paper, matching both topic (NL-to-FOL formalization with
|
|
69
|
+
fine-tuned LLMs) and authorship (the dataset's Hugging Face account is
|
|
70
|
+
``fvossel``): Vossel, Felix, Till Mossakowski, and Bjoern Gehrke. "Advancing
|
|
71
|
+
Natural Language Formalization to First Order Logic with Fine-tuned LLMs."
|
|
72
|
+
arXiv:2509.22338. This link is NOT asserted by the dataset card itself —
|
|
73
|
+
:data:`DATASET_INFO`'s ``citation_hint`` for ``"groves"`` flags it as
|
|
74
|
+
plausible-but-unconfirmed rather than presenting it as settled.
|
|
75
|
+
|
|
76
|
+
What GROVES does NOT have (spelled out so nothing here is silently assumed):
|
|
77
|
+
no premises/entailment structure (a straight translation-pair dataset, like
|
|
78
|
+
MALLS, unlike FOLIO); no entailment label; no id field of any kind; no
|
|
79
|
+
train/validation/test SPLIT FILE bundling — each split is its own upstream
|
|
80
|
+
JSON array file, and this loader (like ``load_malls``) takes exactly one
|
|
81
|
+
already-converted local JSONL path per call, so loading multiple splits
|
|
82
|
+
means calling :func:`load_groves` once per split file; no independent
|
|
83
|
+
human/semantic verification of the FOL column beyond "the dataset was
|
|
84
|
+
filtered" per the card — this adapter's own :func:`~unicode_logic_kit.eval.datasets.audit_examples`
|
|
85
|
+
only checks parseability and well-formedness (closed, arity-consistent,
|
|
86
|
+
lambda-free), never whether a given FOL formula is a *correct* translation
|
|
87
|
+
of its NL sentence.
|
|
88
|
+
|
|
89
|
+
This module never downloads anything — obtain and convert the data yourself
|
|
90
|
+
and pass its local JSONL path to :func:`load_groves`.
|
|
91
|
+
"""
|
|
92
|
+
|
|
93
|
+
import json
|
|
94
|
+
from pathlib import Path
|
|
95
|
+
from typing import FrozenSet, Iterator, Union
|
|
96
|
+
|
|
97
|
+
from ._base import DatasetExample, _register_dataset_info
|
|
98
|
+
|
|
99
|
+
__all__ = ["load_groves"]
|
|
100
|
+
|
|
101
|
+
_register_dataset_info(
|
|
102
|
+
"groves",
|
|
103
|
+
license=(
|
|
104
|
+
"CC-BY-NC-4.0 (non-commercial); GROVES is built from WillowNLtoFOL "
|
|
105
|
+
"(CC BY-NC-ND 4.0, used with permission from its original authors "
|
|
106
|
+
"per the GROVES dataset card) and MALLS-v0 (CC BY-NC 4.0) "
|
|
107
|
+
"natural-language inputs, so downstream use must also honour both "
|
|
108
|
+
"of those source licenses per the dataset card's Licensing section"
|
|
109
|
+
),
|
|
110
|
+
source_url="https://huggingface.co/datasets/fvossel/groves",
|
|
111
|
+
citation_hint=(
|
|
112
|
+
"Vossel, Felix. \"Groves\" (dataset card, Hugging Face, "
|
|
113
|
+
"https://huggingface.co/datasets/fvossel/groves) — the card states "
|
|
114
|
+
"only that \"further details ... will be provided in a forthcoming "
|
|
115
|
+
"publication\" and names no paper. Plausible (topic- and "
|
|
116
|
+
"author-matched, but NOT card-confirmed) companion paper: Vossel, "
|
|
117
|
+
"Felix, Till Mossakowski, and Bjoern Gehrke. \"Advancing Natural "
|
|
118
|
+
"Language Formalization to First Order Logic with Fine-tuned "
|
|
119
|
+
"LLMs.\" arXiv:2509.22338."
|
|
120
|
+
),
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _example_from_record(record: dict, line_no: int,
|
|
125
|
+
known_bad_ids: FrozenSet[str]) -> DatasetExample:
|
|
126
|
+
nl = record.get("NL")
|
|
127
|
+
fol = record.get("FOL")
|
|
128
|
+
# GROVES has no native id field (see module docstring): every verified
|
|
129
|
+
# row is exactly {"NL": ..., "FOL": ...}. Mirrors malls.py's stance —
|
|
130
|
+
# honour an "id" key opportunistically should some downstream re-export
|
|
131
|
+
# add one, otherwise fall back to a positional id.
|
|
132
|
+
raw_id = record.get("id")
|
|
133
|
+
example_id = str(raw_id) if raw_id is not None else f"groves:{line_no}"
|
|
134
|
+
|
|
135
|
+
meta = {k: v for k, v in record.items() if k not in ("NL", "FOL", "id")}
|
|
136
|
+
meta["line_no"] = line_no
|
|
137
|
+
|
|
138
|
+
return DatasetExample(
|
|
139
|
+
id=example_id,
|
|
140
|
+
nl_premises=(),
|
|
141
|
+
fol_premises=(),
|
|
142
|
+
nl_conclusion=nl,
|
|
143
|
+
fol_conclusion=fol,
|
|
144
|
+
label=None,
|
|
145
|
+
known_bad=example_id in known_bad_ids,
|
|
146
|
+
meta=meta,
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def load_groves(path: Union[str, Path], *,
|
|
151
|
+
known_bad_ids: FrozenSet[str] = frozenset()) -> Iterator[DatasetExample]:
|
|
152
|
+
"""Stream :class:`~unicode_logic_kit.eval.datasets.DatasetExample` from a
|
|
153
|
+
local GROVES JSONL file (one already-converted split — see module
|
|
154
|
+
docstring for converting the upstream JSON-array split files).
|
|
155
|
+
|
|
156
|
+
Args:
|
|
157
|
+
path: path to a local ``.jsonl`` file — one ``{"NL": ..., "FOL": ...}``
|
|
158
|
+
object per non-blank line. NEVER downloaded by this function;
|
|
159
|
+
obtain and convert the split file yourself from
|
|
160
|
+
https://huggingface.co/datasets/fvossel/groves.
|
|
161
|
+
known_bad_ids: ids (the record's own ``"id"`` if present, else the
|
|
162
|
+
positional fallback ``f"groves:{line_no}"``) whose ``"FOL"``
|
|
163
|
+
translation is known to be broken. Every yielded example with a
|
|
164
|
+
matching id gets ``known_bad=True``. Defaults to an empty set.
|
|
165
|
+
|
|
166
|
+
Yields:
|
|
167
|
+
One :class:`~unicode_logic_kit.eval.datasets.DatasetExample` per
|
|
168
|
+
non-blank JSONL line, in file order, with ``nl_conclusion``/
|
|
169
|
+
``fol_conclusion`` set from ``"NL"``/``"FOL"`` and
|
|
170
|
+
``nl_premises``/``fol_premises`` empty (see module docstring for why).
|
|
171
|
+
A record missing ``"NL"`` and/or ``"FOL"`` yields an example with
|
|
172
|
+
``None`` in the corresponding field rather than raising — the same
|
|
173
|
+
documented "missing key -> None/() default" behaviour as
|
|
174
|
+
:func:`~unicode_logic_kit.eval.datasets.folio.load_folio` and
|
|
175
|
+
:func:`~unicode_logic_kit.eval.datasets.malls.load_malls` (a structurally
|
|
176
|
+
malformed FILE is a loud failure, see below; a single record missing
|
|
177
|
+
an optional-looking key is not).
|
|
178
|
+
|
|
179
|
+
Raises:
|
|
180
|
+
FileNotFoundError: ``path`` does not exist.
|
|
181
|
+
json.JSONDecodeError: a non-blank line is not valid JSON — raised,
|
|
182
|
+
not swallowed (a malformed dataset file must fail loudly).
|
|
183
|
+
"""
|
|
184
|
+
path = Path(path)
|
|
185
|
+
with path.open("r", encoding="utf-8") as fh:
|
|
186
|
+
for line_no, raw_line in enumerate(fh):
|
|
187
|
+
line = raw_line.strip()
|
|
188
|
+
if not line:
|
|
189
|
+
continue
|
|
190
|
+
record = json.loads(line)
|
|
191
|
+
yield _example_from_record(record, line_no, known_bad_ids)
|
|
@@ -0,0 +1,467 @@
|
|
|
1
|
+
"""Adapter for LogicBench (Parmar, Patel, Varshney, Nakamura, Luo, Mashetty,
|
|
2
|
+
Mitra, Baral, "Towards Systematic Evaluation of Logical Reasoning Ability of
|
|
3
|
+
Large Language Models", 2024, arXiv:2404.15522) — natural-language
|
|
4
|
+
question-answering over 25 single-inference-rule reasoning patterns spanning
|
|
5
|
+
propositional, first-order and non-monotonic logic. Modeled directly on
|
|
6
|
+
:mod:`~unicode_logic_kit.eval.datasets.fracas`, since LogicBench ships NO gold
|
|
7
|
+
FOL either: every row is natural-language context plus a question and a
|
|
8
|
+
yes/no or multiple-choice answer, nothing more — so, exactly as for FraCaS,
|
|
9
|
+
the translation step lives OUTSIDE this library (:func:`solve_example`'s
|
|
10
|
+
``translate``) and this package only decides.
|
|
11
|
+
|
|
12
|
+
Source and verified schema
|
|
13
|
+
---------------------------
|
|
14
|
+
Verified 2026-09-17 directly against a local clone of the repository the
|
|
15
|
+
paper names, ``https://github.com/Mihir3009/LogicBench``. ``data/`` holds
|
|
16
|
+
two releases:
|
|
17
|
+
|
|
18
|
+
* **LogicBench(Eval)** — the human-verified evaluation set this adapter
|
|
19
|
+
reads, split into ``BQA`` (Binary Question-Answering) and ``MCQA``
|
|
20
|
+
(Multiple-Choice Question-Answering), each further split by
|
|
21
|
+
``propositional_logic`` / ``first_order_logic`` / ``nm_logic`` and then by
|
|
22
|
+
one JSON file per inference rule ("axiom"), e.g.
|
|
23
|
+
``BQA/propositional_logic/modus_tollens/data_instances.json``.
|
|
24
|
+
* **LogicBench(Aug)** — a synthetically augmented TRAINING split with a
|
|
25
|
+
DIFFERENT schema (top-level key ``"data_samples"`` instead of
|
|
26
|
+
``"samples"``, no per-row ``"id"``, and 4 ``qa_pairs`` per row — both
|
|
27
|
+
polarities of both directions — instead of 2). Inspected while verifying
|
|
28
|
+
this adapter's schema, but NOT read by it: the build spec this module
|
|
29
|
+
implements only covers ``LogicBench(Eval)``, and Aug's different shape
|
|
30
|
+
would need its own field mapping, not this one.
|
|
31
|
+
|
|
32
|
+
One **Eval** JSON file — :func:`load_logicbench` reads exactly one, like
|
|
33
|
+
every other loader in this package; nothing here downloads anything — has
|
|
34
|
+
the shape ``{"type": str, "axiom": str, "samples": [...]}``:
|
|
35
|
+
|
|
36
|
+
* ``"type"`` — the file's own logic-type label, kept VERBATIM in
|
|
37
|
+
``meta["logic_type"]``. Confirmed by reading every real Eval file: it is
|
|
38
|
+
``"propositional_logic"`` / ``"first_order_logic"`` under those two
|
|
39
|
+
directories (matching the directory name), but ``"non_monotonic_logic"``
|
|
40
|
+
under the ``nm_logic`` DIRECTORY — never the literal string
|
|
41
|
+
``"nm_logic"``. :func:`solve_example` routes on the ACTUAL file value
|
|
42
|
+
(``"non_monotonic_logic"``, :data:`NM_LOGIC_TYPE`), not on the directory
|
|
43
|
+
name a caller happened to read the file from.
|
|
44
|
+
* ``"axiom"`` — the inference-rule name (e.g. ``"modus_tollens"``), kept
|
|
45
|
+
verbatim in ``meta["axiom"]``.
|
|
46
|
+
* ``"samples"`` — a list of ``{"id": int, "context": str, ...}``. ``id`` is
|
|
47
|
+
1-based WITHIN THIS ONE FILE, not globally unique (kept verbatim in
|
|
48
|
+
``meta["sample_id"]``); ``context`` is one NL paragraph, never pre-split
|
|
49
|
+
into sentences the way FraCaS's ``<p>`` elements are, so it maps to the
|
|
50
|
+
single-element ``nl_premises = (context,)`` — a caller's ``translate``
|
|
51
|
+
sees the whole paragraph at once and is free to return a single
|
|
52
|
+
conjunctive formula for it.
|
|
53
|
+
|
|
54
|
+
* **BQA** (``split="BQA"``): each sample additionally carries
|
|
55
|
+
``"qa_pairs": [{"question": str, "answer": "yes"|"no"}, ...]`` — 2 to 4
|
|
56
|
+
pairs per sample in every real Eval BQA file (every ``answer`` verified
|
|
57
|
+
to be exactly the lowercase string ``"yes"`` or ``"no"``, nothing else).
|
|
58
|
+
A sample with N qa_pairs yields N separate :class:`DatasetExample`\\ s
|
|
59
|
+
(one per question, all sharing the same ``nl_premises``), since each
|
|
60
|
+
pair asks about a DIFFERENT proposition with its OWN gold answer —
|
|
61
|
+
collapsing them into one example would silently keep only one label.
|
|
62
|
+
``meta["qa_index"]`` (0-based, within the sample) makes the split point
|
|
63
|
+
reconstructable, and the synthetic id embeds it too.
|
|
64
|
+
* **MCQA** (``split="MCQA"``): each sample instead carries
|
|
65
|
+
``"question": str`` (a FIXED meta-question, e.g. "What would be the most
|
|
66
|
+
appropriate conclusion based on the given context?" — not itself a
|
|
67
|
+
provable proposition; see :func:`solve_example`), ``"choices": dict``
|
|
68
|
+
(``"choice_1"``, … — 4 or 5 entries depending on the file, verified) and
|
|
69
|
+
``"answer": str`` (verified, in every real Eval MCQA file, to always be
|
|
70
|
+
a key of that SAME sample's ``choices``). One :class:`DatasetExample`
|
|
71
|
+
per sample. ``choices`` survives verbatim in ``meta["choices"]``.
|
|
72
|
+
|
|
73
|
+
Field mapping (either split): ``nl_premises = (context,)``, ``nl_conclusion``
|
|
74
|
+
= the question text, ``label`` = the answer — ``"yes"``/``"no"`` verbatim
|
|
75
|
+
for BQA, the chosen ``"choice_N"`` key verbatim for MCQA. Kept AS-IS, not
|
|
76
|
+
smoothed into FraCaS's yes/no/unknown three-way scale: a binary
|
|
77
|
+
question-answering task and a 4-or-5-way multiple choice are different task
|
|
78
|
+
shapes, and forcing one vocabulary onto both would invent structure that is
|
|
79
|
+
not in the data. ``fol_premises = ()`` and ``fol_conclusion = None``
|
|
80
|
+
always — see "Honest limitations".
|
|
81
|
+
|
|
82
|
+
Honest limitations
|
|
83
|
+
-------------------
|
|
84
|
+
* No gold FOL anywhere in the source, so — exactly as for
|
|
85
|
+
:mod:`~unicode_logic_kit.eval.datasets.fracas` —
|
|
86
|
+
:func:`~unicode_logic_kit.eval.datasets.audit_examples` is vacuous on every
|
|
87
|
+
LogicBench example (nothing to audit), and deciding one needs an
|
|
88
|
+
externally injected translation.
|
|
89
|
+
* **MCQA rows are not decided by this module at all.** ``"question"`` in an
|
|
90
|
+
MCQA sample is a generic meta-question ("What would be the most
|
|
91
|
+
appropriate conclusion...?"), not a standalone proposition — there is
|
|
92
|
+
nothing there for a prover to prove or refute. :func:`solve_example`
|
|
93
|
+
refuses every MCQA row by name rather than silently running
|
|
94
|
+
``api.prove`` on a sentence that was never meant to be one; a caller who
|
|
95
|
+
wants to score MCQA has to translate one of ``example.meta["choices"]``
|
|
96
|
+
itself and decide it directly. Consequently the DISTRACTOR choices are
|
|
97
|
+
never logic-checked by this adapter at all, chosen or not.
|
|
98
|
+
* **The non-monotonic route is a real, narrow fragment, not a general
|
|
99
|
+
solver.** :mod:`~unicode_logic_kit.semantics.nonmonotonic`'s
|
|
100
|
+
``minimal_models``/``minimal_entails`` implement circumscription with
|
|
101
|
+
every predicate either CIRCUMSCRIBED (minimised) or FIXED — there is no
|
|
102
|
+
third "varied" category (that module's own ``circumscription_formula``
|
|
103
|
+
explicitly leaves it unimplemented). A translation that leaves an
|
|
104
|
+
"abnormality" predicate free for some individual (rather than pinning it
|
|
105
|
+
with an explicit ground fact, positive or negative) will generally admit
|
|
106
|
+
several incomparable minimal models that DISAGREE on the goal.
|
|
107
|
+
:func:`solve_example` does **not** detect that disagreement and does
|
|
108
|
+
**not** refuse it: ``minimal_entails`` only ever returns a bool, with no
|
|
109
|
+
way to report that its minimal models disagree, so the route falls
|
|
110
|
+
through to that bool's own SKEPTICAL reading — "yes" iff the goal holds
|
|
111
|
+
in *every* minimal model found, "no" otherwise — and reports it exactly
|
|
112
|
+
like any other answer, with no flag that several readings were possible.
|
|
113
|
+
That skeptical bool is a real, well-defined answer to a real, precisely
|
|
114
|
+
bounded question (minimal-model entailment up to ``max_size``), not a
|
|
115
|
+
guess or an approximation of one — but it is only as trustworthy as the
|
|
116
|
+
translation's discipline in pinning every abnormality predicate with an
|
|
117
|
+
explicit ground fact for every named individual; a translation that
|
|
118
|
+
skips one gets a confident-looking "yes"/"no" out of this route with no
|
|
119
|
+
signal that the question was underspecified. The ONE case this route
|
|
120
|
+
does detect and refuse is the *empty* minimal-model set (see
|
|
121
|
+
:func:`solve_example`'s own docstring) — genuinely unsatisfiable
|
|
122
|
+
premises, or a search bound that is simply too small.
|
|
123
|
+
|
|
124
|
+
License
|
|
125
|
+
-------
|
|
126
|
+
The roadmap build spec that requested this adapter named CC BY 4.0. That is
|
|
127
|
+
WRONG for this repository: the cloned ``LogicBench`` repository's
|
|
128
|
+
``LICENSE`` file is the plain MIT License (``Copyright (c) 2024 Mihir``),
|
|
129
|
+
and its ``README.md`` states "**Licence:** MIT License" directly under its
|
|
130
|
+
"Data Release" heading — both read directly, 2026-09-17.
|
|
131
|
+
:data:`~unicode_logic_kit.eval.datasets.DATASET_INFO` records MIT, not the
|
|
132
|
+
spec's CC BY 4.0.
|
|
133
|
+
"""
|
|
134
|
+
|
|
135
|
+
import json
|
|
136
|
+
from pathlib import Path
|
|
137
|
+
from typing import Callable, FrozenSet, Iterator, Optional, Set, Union
|
|
138
|
+
|
|
139
|
+
from ._base import DatasetExample, _register_dataset_info
|
|
140
|
+
from ...semantics.modelfinder import MAX_CANDIDATES
|
|
141
|
+
|
|
142
|
+
__all__ = ["load_logicbench", "solve_example", "LOGIC_TYPES", "NM_LOGIC_TYPE"]
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
#: The three ``"type"`` values a real LogicBench(Eval) file carries — see the
|
|
146
|
+
#: module docstring for why the non-monotonic one is NOT the string
|
|
147
|
+
#: ``"nm_logic"`` despite that being the directory name upstream.
|
|
148
|
+
LOGIC_TYPES = ("propositional_logic", "first_order_logic", "non_monotonic_logic")
|
|
149
|
+
|
|
150
|
+
#: The ``"type"`` value :func:`solve_example` routes through
|
|
151
|
+
#: :mod:`~unicode_logic_kit.semantics.nonmonotonic` instead of ``api.prove``.
|
|
152
|
+
NM_LOGIC_TYPE = "non_monotonic_logic"
|
|
153
|
+
|
|
154
|
+
_CLASSICAL_LOGIC_TYPES = frozenset(LOGIC_TYPES) - {NM_LOGIC_TYPE}
|
|
155
|
+
_SPLITS = ("BQA", "MCQA")
|
|
156
|
+
_BQA_ANSWERS = frozenset({"yes", "no"})
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
_register_dataset_info(
|
|
160
|
+
"logicbench",
|
|
161
|
+
license=("MIT License (the repository's own LICENSE file and its "
|
|
162
|
+
"README's 'Licence: MIT License' agree; verified 2026-09-17 — "
|
|
163
|
+
"NOT CC BY 4.0, which is what an earlier, unverified roadmap "
|
|
164
|
+
"entry for this adapter had assumed)"),
|
|
165
|
+
source_url="https://github.com/Mihir3009/LogicBench",
|
|
166
|
+
citation_hint=('Parmar, Patel, Varshney, Nakamura, Luo, Mashetty, Mitra, '
|
|
167
|
+
'Baral, "Towards Systematic Evaluation of Logical '
|
|
168
|
+
'Reasoning Ability of Large Language Models", 2024, '
|
|
169
|
+
'arXiv:2404.15522.'),
|
|
170
|
+
)
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
# ---------------------------------------------------------------------------
|
|
174
|
+
# Reading
|
|
175
|
+
# ---------------------------------------------------------------------------
|
|
176
|
+
|
|
177
|
+
def load_logicbench(path: Union[str, Path], *, split: str,
|
|
178
|
+
known_bad_ids: FrozenSet[str] = frozenset(),
|
|
179
|
+
) -> Iterator[DatasetExample]:
|
|
180
|
+
"""Read one LogicBench(Eval) ``data_instances.json`` into
|
|
181
|
+
:class:`DatasetExample` objects, in file order.
|
|
182
|
+
|
|
183
|
+
See the module docstring for the field mapping and the BQA
|
|
184
|
+
qa_pairs-flattening this loader does. ``split`` must be ``"BQA"`` or
|
|
185
|
+
``"MCQA"`` and must match the file's ACTUAL sample shape — a BQA sample
|
|
186
|
+
without ``qa_pairs`` (or an MCQA sample without ``question``/``choices``)
|
|
187
|
+
is refused by name rather than silently misread, so passing the wrong
|
|
188
|
+
``split`` for a file cannot produce garbage examples.
|
|
189
|
+
|
|
190
|
+
Args:
|
|
191
|
+
path: the local ``data_instances.json`` (one axiom, one logic type,
|
|
192
|
+
one of BQA/MCQA).
|
|
193
|
+
split: ``"BQA"`` or ``"MCQA"`` — which of LogicBench(Eval)'s two
|
|
194
|
+
task shapes this file holds.
|
|
195
|
+
known_bad_ids: ids (in this adapter's own prefixed form, e.g.
|
|
196
|
+
``"logicbench:BQA:propositional_logic:modus_tollens:1:0"``) to
|
|
197
|
+
flag as ``known_bad`` — the same caller-curated mechanic every
|
|
198
|
+
adapter has.
|
|
199
|
+
|
|
200
|
+
Raises:
|
|
201
|
+
ValueError: ``split`` is not one of ``"BQA"``/``"MCQA"``, the file
|
|
202
|
+
is missing ``type``/``axiom``/``samples``, ``type`` is outside
|
|
203
|
+
:data:`LOGIC_TYPES`, a sample id repeats, a BQA sample has no
|
|
204
|
+
``qa_pairs`` (or an answer outside ``{"yes", "no"}``), or an
|
|
205
|
+
MCQA sample has no ``question``/``choices`` (or an ``answer``
|
|
206
|
+
that is not itself a key of its own ``choices``) — every
|
|
207
|
+
malformed-input case is named, never worked around.
|
|
208
|
+
"""
|
|
209
|
+
if split not in _SPLITS:
|
|
210
|
+
raise ValueError(
|
|
211
|
+
f"logicbench: split must be one of {_SPLITS}, got {split!r}")
|
|
212
|
+
|
|
213
|
+
with open(path, encoding="utf-8") as f:
|
|
214
|
+
data = json.load(f)
|
|
215
|
+
|
|
216
|
+
if not isinstance(data, dict) or not {"type", "axiom", "samples"} <= data.keys():
|
|
217
|
+
raise ValueError(
|
|
218
|
+
f"logicbench: {path} is missing 'type'/'axiom'/'samples' — is "
|
|
219
|
+
"this a LogicBench(Eval) data_instances.json?")
|
|
220
|
+
|
|
221
|
+
logic_type = data["type"]
|
|
222
|
+
if logic_type not in LOGIC_TYPES:
|
|
223
|
+
raise ValueError(
|
|
224
|
+
f"logicbench: {path} has type={logic_type!r}, outside "
|
|
225
|
+
f"{list(LOGIC_TYPES)}")
|
|
226
|
+
axiom = data["axiom"]
|
|
227
|
+
if not axiom:
|
|
228
|
+
raise ValueError(f"logicbench: {path} has an empty 'axiom'")
|
|
229
|
+
|
|
230
|
+
samples = data["samples"]
|
|
231
|
+
if not isinstance(samples, list):
|
|
232
|
+
raise ValueError(f"logicbench: {path}: 'samples' is not a list")
|
|
233
|
+
|
|
234
|
+
seen_sample_ids: Set = set()
|
|
235
|
+
for sample in samples:
|
|
236
|
+
sample_id = sample.get("id")
|
|
237
|
+
if sample_id is None:
|
|
238
|
+
raise ValueError(f"logicbench: {path}: a sample has no 'id'")
|
|
239
|
+
if sample_id in seen_sample_ids:
|
|
240
|
+
raise ValueError(
|
|
241
|
+
f"logicbench: {path}: duplicate sample id {sample_id!r}")
|
|
242
|
+
seen_sample_ids.add(sample_id)
|
|
243
|
+
|
|
244
|
+
context = sample.get("context")
|
|
245
|
+
if not context:
|
|
246
|
+
raise ValueError(
|
|
247
|
+
f"logicbench: {path}: sample {sample_id} has an empty "
|
|
248
|
+
"'context'")
|
|
249
|
+
|
|
250
|
+
base_meta = {"axiom": axiom, "logic_type": logic_type, "task": split,
|
|
251
|
+
"sample_id": sample_id}
|
|
252
|
+
|
|
253
|
+
if split == "BQA":
|
|
254
|
+
qa_pairs = sample.get("qa_pairs")
|
|
255
|
+
if not qa_pairs:
|
|
256
|
+
raise ValueError(
|
|
257
|
+
f"logicbench: {path}: sample {sample_id} has no "
|
|
258
|
+
"'qa_pairs' — is split='BQA' correct for this file?")
|
|
259
|
+
for qa_index, qa in enumerate(qa_pairs):
|
|
260
|
+
question = qa.get("question")
|
|
261
|
+
answer = qa.get("answer")
|
|
262
|
+
if not question:
|
|
263
|
+
raise ValueError(
|
|
264
|
+
f"logicbench: {path}: sample {sample_id} "
|
|
265
|
+
f"qa_pairs[{qa_index}] has no 'question'")
|
|
266
|
+
if answer not in _BQA_ANSWERS:
|
|
267
|
+
raise ValueError(
|
|
268
|
+
f"logicbench: {path}: sample {sample_id} "
|
|
269
|
+
f"qa_pairs[{qa_index}] has answer={answer!r}, "
|
|
270
|
+
f"outside {sorted(_BQA_ANSWERS)}")
|
|
271
|
+
example_id = (f"logicbench:BQA:{logic_type}:{axiom}:"
|
|
272
|
+
f"{sample_id}:{qa_index}")
|
|
273
|
+
yield DatasetExample(
|
|
274
|
+
id=example_id,
|
|
275
|
+
nl_premises=(context,),
|
|
276
|
+
fol_premises=(),
|
|
277
|
+
nl_conclusion=question,
|
|
278
|
+
fol_conclusion=None,
|
|
279
|
+
label=answer,
|
|
280
|
+
known_bad=example_id in known_bad_ids,
|
|
281
|
+
meta=dict(base_meta, qa_index=qa_index),
|
|
282
|
+
)
|
|
283
|
+
else: # "MCQA"
|
|
284
|
+
question = sample.get("question")
|
|
285
|
+
choices = sample.get("choices")
|
|
286
|
+
answer = sample.get("answer")
|
|
287
|
+
if not question:
|
|
288
|
+
raise ValueError(
|
|
289
|
+
f"logicbench: {path}: sample {sample_id} has no "
|
|
290
|
+
"'question' — is split='MCQA' correct for this file?")
|
|
291
|
+
if not isinstance(choices, dict) or not choices:
|
|
292
|
+
raise ValueError(
|
|
293
|
+
f"logicbench: {path}: sample {sample_id} has no "
|
|
294
|
+
"'choices' dict")
|
|
295
|
+
if answer not in choices:
|
|
296
|
+
raise ValueError(
|
|
297
|
+
f"logicbench: {path}: sample {sample_id} has "
|
|
298
|
+
f"answer={answer!r}, not a key of its own 'choices' "
|
|
299
|
+
f"{sorted(choices)}")
|
|
300
|
+
example_id = f"logicbench:MCQA:{logic_type}:{axiom}:{sample_id}"
|
|
301
|
+
yield DatasetExample(
|
|
302
|
+
id=example_id,
|
|
303
|
+
nl_premises=(context,),
|
|
304
|
+
fol_premises=(),
|
|
305
|
+
nl_conclusion=question,
|
|
306
|
+
fol_conclusion=None,
|
|
307
|
+
label=answer,
|
|
308
|
+
known_bad=example_id in known_bad_ids,
|
|
309
|
+
meta=dict(base_meta, choices=dict(choices)),
|
|
310
|
+
)
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
# ---------------------------------------------------------------------------
|
|
314
|
+
# Deciding — with the translation injected by the caller
|
|
315
|
+
# ---------------------------------------------------------------------------
|
|
316
|
+
|
|
317
|
+
def solve_example(example: DatasetExample, *, translate: Callable[[str], object],
|
|
318
|
+
circumscribed: Optional[Set] = None,
|
|
319
|
+
max_size: int = 4, max_candidates: int = MAX_CANDIDATES,
|
|
320
|
+
**prove_kwargs) -> dict:
|
|
321
|
+
"""Decide one LogicBench BQA row end-to-end — the translation is YOURS.
|
|
322
|
+
|
|
323
|
+
LogicBench ships no formulas, so this helper takes ``translate``: a
|
|
324
|
+
callable mapping one natural-language sentence to either a formula
|
|
325
|
+
string (parsed with :func:`unicode_logic_kit.api.parse_any`) or an
|
|
326
|
+
already-built kit node — the same seam
|
|
327
|
+
:func:`~unicode_logic_kit.eval.datasets.fracas.solve_example` uses. This
|
|
328
|
+
package calls no LLM and no external system itself.
|
|
329
|
+
|
|
330
|
+
Routing is on ``example.meta["logic_type"]``:
|
|
331
|
+
|
|
332
|
+
* :data:`NM_LOGIC_TYPE` (``"non_monotonic_logic"``): decided via
|
|
333
|
+
:func:`unicode_logic_kit.semantics.nonmonotonic.minimal_entails` —
|
|
334
|
+
``circumscribed`` (``None`` minimises every predicate in the
|
|
335
|
+
translated theory, matching that module's own default reading),
|
|
336
|
+
``max_size`` and ``max_candidates`` reach it verbatim. Before trusting
|
|
337
|
+
the answer, this function ALSO calls
|
|
338
|
+
:func:`~unicode_logic_kit.semantics.nonmonotonic.minimal_models`
|
|
339
|
+
directly and checks it is non-empty: ``minimal_entails`` returns
|
|
340
|
+
``True`` VACUOUSLY when no minimal model exists within the bound
|
|
341
|
+
(nothing to check the conclusion against — for-loop over an empty
|
|
342
|
+
list), which would silently misreport either genuinely unsatisfiable
|
|
343
|
+
premises or a search bound that is simply too small as a confident
|
|
344
|
+
"yes". Finding no minimal model raises instead, naming the reason,
|
|
345
|
+
rather than ever returning that vacuous "yes". This is the ONLY
|
|
346
|
+
case this route detects and refuses: a translation that leaves an
|
|
347
|
+
abnormality predicate's value free for some individual (instead of
|
|
348
|
+
pinning it, positive or negative, with an explicit ground fact) will
|
|
349
|
+
typically produce several incomparable minimal models that DISAGREE
|
|
350
|
+
on the goal rather than an empty set, so it is NOT caught here —
|
|
351
|
+
:func:`minimal_entails` only ever returns a bool, with no way to
|
|
352
|
+
report that its minimal models disagree, so this function silently
|
|
353
|
+
reports that bool's own skeptical reading ("yes" iff the goal holds
|
|
354
|
+
in every minimal model found) with no signal that several readings
|
|
355
|
+
were possible. See the module docstring's "Honest limitations"
|
|
356
|
+
section for the discipline a translation needs (an explicit ground
|
|
357
|
+
fact for every abnormality predicate application) to avoid that
|
|
358
|
+
silent case.
|
|
359
|
+
* :data:`~unicode_logic_kit.eval.datasets.logicbench.LOGIC_TYPES`'s other
|
|
360
|
+
two values (``"propositional_logic"``, ``"first_order_logic"``):
|
|
361
|
+
decided via :func:`unicode_logic_kit.api.prove` — ``"yes"`` iff PROVED,
|
|
362
|
+
``"no"`` iff REFUTED (a genuine countermodel, not merely "could not
|
|
363
|
+
prove"), and ``predicted=None`` on an inconclusive UNKNOWN verdict:
|
|
364
|
+
LogicBench's label vocabulary is only ``{"yes", "no"}``, so — unlike
|
|
365
|
+
FraCaS, whose own three-way scale has an ``"unknown"`` label to fall
|
|
366
|
+
back on — forcing an indefinite prover outcome into either binary
|
|
367
|
+
label would invent an answer LogicBench never asked for. Extra
|
|
368
|
+
``prove_kwargs`` reach :func:`unicode_logic_kit.api.prove` verbatim (and
|
|
369
|
+
are IGNORED on a non-monotonic-logic row — that route takes
|
|
370
|
+
``circumscribed``/``max_size``/``max_candidates`` instead).
|
|
371
|
+
|
|
372
|
+
Only decides BQA rows. An MCQA row's ``nl_conclusion`` is a generic
|
|
373
|
+
meta-question, not a standalone proposition (see the module docstring),
|
|
374
|
+
so this function refuses it by name instead of running a prover on a
|
|
375
|
+
sentence that was never meant to be one.
|
|
376
|
+
|
|
377
|
+
Returns a dict with ``predicted`` (``"yes"``/``"no"``/``None``),
|
|
378
|
+
``label`` (the gold answer, untouched — scoring against it is the
|
|
379
|
+
caller's decision), ``route`` (``"classical"``/``"nonmonotonic"``), the
|
|
380
|
+
translated ``premises``/``hypothesis`` in kit notation (so a wrong
|
|
381
|
+
prediction can be traced back to the translation that caused it), and
|
|
382
|
+
either ``verdict`` (the classical route's full
|
|
383
|
+
:class:`~unicode_logic_kit.atp.protocol.Verdict` dict) or
|
|
384
|
+
``minimal_model_count`` (the non-monotonic route's model count).
|
|
385
|
+
|
|
386
|
+
Raises:
|
|
387
|
+
ValueError: ``example`` is an MCQA row, its ``logic_type`` is
|
|
388
|
+
outside :data:`LOGIC_TYPES`, a translated string does not parse,
|
|
389
|
+
or (non-monotonic route only) no minimal model of the
|
|
390
|
+
translated premises was found within ``max_size``.
|
|
391
|
+
"""
|
|
392
|
+
from ... import api
|
|
393
|
+
from ...fol.nodes import Node
|
|
394
|
+
from ...semantics.nonmonotonic import minimal_entails, minimal_models
|
|
395
|
+
|
|
396
|
+
if example.meta.get("task") != "BQA":
|
|
397
|
+
raise ValueError(
|
|
398
|
+
f"logicbench: example {example.id}: solve_example only decides "
|
|
399
|
+
"BQA rows — an MCQA row's 'question' field is a generic "
|
|
400
|
+
"meta-question ('What would be the most appropriate "
|
|
401
|
+
"conclusion...?'), not a standalone provable proposition; "
|
|
402
|
+
"translate one of example.meta['choices'] yourself and decide "
|
|
403
|
+
"it directly instead.")
|
|
404
|
+
|
|
405
|
+
def _formula(sentence: str) -> "Node":
|
|
406
|
+
produced = translate(sentence)
|
|
407
|
+
if isinstance(produced, Node):
|
|
408
|
+
return produced
|
|
409
|
+
if not isinstance(produced, str):
|
|
410
|
+
raise ValueError(
|
|
411
|
+
f"logicbench: example {example.id}: translate({sentence!r}) "
|
|
412
|
+
f"returned {type(produced).__name__}, expected a formula "
|
|
413
|
+
"string or a kit node")
|
|
414
|
+
parsed = api.parse_any(produced)
|
|
415
|
+
if not parsed.ok:
|
|
416
|
+
raise ValueError(
|
|
417
|
+
f"logicbench: example {example.id}: the translation "
|
|
418
|
+
f"{produced!r} of {sentence!r} does not parse")
|
|
419
|
+
return parsed.formula
|
|
420
|
+
|
|
421
|
+
premises = [_formula(sentence) for sentence in example.nl_premises]
|
|
422
|
+
hypothesis = _formula(example.nl_conclusion)
|
|
423
|
+
|
|
424
|
+
result = {
|
|
425
|
+
"label": example.label,
|
|
426
|
+
"premises": [p.to_unicode_str() for p in premises],
|
|
427
|
+
"hypothesis": hypothesis.to_unicode_str(),
|
|
428
|
+
}
|
|
429
|
+
|
|
430
|
+
logic_type = example.meta.get("logic_type")
|
|
431
|
+
if logic_type == NM_LOGIC_TYPE:
|
|
432
|
+
found = minimal_models(premises, circumscribed, max_size=max_size,
|
|
433
|
+
max_candidates=max_candidates,
|
|
434
|
+
extra_signature=[hypothesis])
|
|
435
|
+
if not found:
|
|
436
|
+
raise ValueError(
|
|
437
|
+
f"logicbench: example {example.id}: "
|
|
438
|
+
"semantics.nonmonotonic found NO minimal model of the "
|
|
439
|
+
f"translated premises within max_size={max_size} — this "
|
|
440
|
+
"route refuses rather than trust minimal_entails's vacuous "
|
|
441
|
+
"'True' for an empty model set (which could mean the "
|
|
442
|
+
"premises are unsatisfiable, or just that the bound is too "
|
|
443
|
+
"small to tell). Widen max_size, or check the translation.")
|
|
444
|
+
entailed = minimal_entails(premises, hypothesis, circumscribed,
|
|
445
|
+
max_size=max_size,
|
|
446
|
+
max_candidates=max_candidates)
|
|
447
|
+
result.update(predicted=("yes" if entailed else "no"),
|
|
448
|
+
route="nonmonotonic", minimal_model_count=len(found))
|
|
449
|
+
return result
|
|
450
|
+
|
|
451
|
+
if logic_type not in _CLASSICAL_LOGIC_TYPES:
|
|
452
|
+
raise ValueError(
|
|
453
|
+
f"logicbench: example {example.id}: unsupported "
|
|
454
|
+
f"logic_type {logic_type!r} — solve_example decides "
|
|
455
|
+
f"{sorted(_CLASSICAL_LOGIC_TYPES)} via api.prove and "
|
|
456
|
+
f"{NM_LOGIC_TYPE!r} via semantics.nonmonotonic, nothing else.")
|
|
457
|
+
|
|
458
|
+
verdict = api.prove(hypothesis, premises, **prove_kwargs)
|
|
459
|
+
if verdict.status == "proved":
|
|
460
|
+
predicted: Optional[str] = "yes"
|
|
461
|
+
elif verdict.status == "refuted":
|
|
462
|
+
predicted = "no"
|
|
463
|
+
else: # "unknown"
|
|
464
|
+
predicted = None
|
|
465
|
+
result.update(predicted=predicted, route="classical",
|
|
466
|
+
verdict=verdict.to_dict())
|
|
467
|
+
return result
|