unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
"""Declared converse / argument-permutation axioms — an opt-in equivalence bridge.
|
|
2
|
+
|
|
3
|
+
Two predicates can denote the same relation with their arguments swapped —
|
|
4
|
+
``LovedBy(x, y)`` and ``Loves(y, x)`` — or, more generally, permuted for any
|
|
5
|
+
arity: ``Between(a, b, c)`` and ``BetweenRev(c, b, a)``. Neither
|
|
6
|
+
:func:`~unicode_logic_kit.eval.canonical.exact_match` (which never renames a
|
|
7
|
+
predicate) nor :func:`~unicode_logic_kit.eval.predicate_match.align_symbols`
|
|
8
|
+
(which renames predicate *names* but never touches argument *order* — see
|
|
9
|
+
that module's docstring) closes this gap, and deliberately so: automatically
|
|
10
|
+
GUESSING that ``LovedBy``/``Loves`` are converses from lexical similarity
|
|
11
|
+
alone is undecidable-from-the-AST and would just as happily "forgive" a
|
|
12
|
+
genuine subject/object-swap translation ERROR (see roadmap item C32's
|
|
13
|
+
rejection). This module takes the opposite approach: the CALLER declares the
|
|
14
|
+
bridge explicitly, per comparison, and only the solver level of
|
|
15
|
+
:func:`unicode_logic_kit.eval.equivalence.equivalent` ever consumes it — see
|
|
16
|
+
that module for how ``converses=`` is threaded through and tagged with its
|
|
17
|
+
own ``method_used`` value (``"solver_modulo_converses"``), never merged into
|
|
18
|
+
plain equivalence.
|
|
19
|
+
|
|
20
|
+
A declaration is a plain, JSON-friendly 3-tuple::
|
|
21
|
+
|
|
22
|
+
(a_key, b_key, permutation)
|
|
23
|
+
|
|
24
|
+
where ``a_key`` / ``b_key`` are ``(name, arity)`` predicate keys — the same
|
|
25
|
+
convention :mod:`~unicode_logic_kit.eval.predicate_match` uses for its own
|
|
26
|
+
symbol inventories — and ``permutation`` is a tuple of ``arity`` distinct
|
|
27
|
+
indices into ``range(arity)``. :func:`converse_axioms` turns a list of these
|
|
28
|
+
into one closed biconditional sentence per declaration:
|
|
29
|
+
|
|
30
|
+
∀v0 … v_{n-1} (A(v0, …, v_{n-1}) ↔ B(v_perm[0], …, v_perm[n-1]))
|
|
31
|
+
|
|
32
|
+
so ``(("LovedBy", 2), ("Loves", 2), (1, 0))`` builds
|
|
33
|
+
``∀v0 ∀v1 (LovedBy(v0, v1) ↔ Loves(v1, v0))``. :func:`validate_converses`
|
|
34
|
+
runs the structural sanity checks (arity match, permutation validity,
|
|
35
|
+
no self-pair, no built-in predicate, no duplicate/contradictory pair) that
|
|
36
|
+
:func:`converse_axioms` also runs internally before building anything, so a
|
|
37
|
+
malformed declaration is refused with :class:`ValueError` before any Z3 call
|
|
38
|
+
is even attempted.
|
|
39
|
+
|
|
40
|
+
On the Z3-domain-sort question a caller might reasonably worry about: this
|
|
41
|
+
kit's whole classical export (:class:`~unicode_logic_kit.fol.nodes.Z3Env`) uses
|
|
42
|
+
exactly ONE Z3 sort for every term, always (``_SORT = z3.DeclareSort("S")`` in
|
|
43
|
+
``fol/_fol_nodes.py`` — confirmed by inspection, the only ``DeclareSort`` call
|
|
44
|
+
in the codebase), and a predicate is interned purely by ``(name, arity)`` —
|
|
45
|
+
its Z3 domain is always ``[_SORT] * arity``, regardless of whether the atom
|
|
46
|
+
sits under a plain :class:`~unicode_logic_kit.fol.nodes.Quantifier` or under a
|
|
47
|
+
:class:`~unicode_logic_kit.fol.nodes.SortedQuantifier` (which auto-relativises
|
|
48
|
+
to plain FOL over the SAME single sort, guarded by a unary sort-membership
|
|
49
|
+
predicate — see ``fol._msfl_nodes.to_fol``). So a plain, unsorted axiom built
|
|
50
|
+
here for a predicate pair ``(name, arity)`` interns to the IDENTICAL Z3
|
|
51
|
+
function declaration the compared formulas themselves use for that pair,
|
|
52
|
+
many-sorted or not — there is no second Z3 domain-sort family for it to
|
|
53
|
+
silently diverge into. See :mod:`unicode_logic_kit.eval.equivalence`'s
|
|
54
|
+
docstring for the differential test that proves this bridges correctly under
|
|
55
|
+
:class:`~unicode_logic_kit.fol.nodes.SortedQuantifier` input, not just for
|
|
56
|
+
unsorted formulas.
|
|
57
|
+
"""
|
|
58
|
+
|
|
59
|
+
from typing import Sequence, Tuple
|
|
60
|
+
|
|
61
|
+
from unicode_logic_kit.fol.nodes import Node, Atom, Iff, Quantifier, Variable
|
|
62
|
+
from .validate import _BUILTIN_PREDS
|
|
63
|
+
|
|
64
|
+
__all__ = ["ConverseDeclaration", "validate_converses", "converse_axioms"]
|
|
65
|
+
|
|
66
|
+
#: A predicate symbol key: ``(name, arity)`` — the same convention
|
|
67
|
+
#: :data:`unicode_logic_kit.eval.predicate_match._SymKey` uses.
|
|
68
|
+
_SymKey = Tuple[str, int]
|
|
69
|
+
|
|
70
|
+
#: One declared converse: ``A`` (natural argument order) is biconditional
|
|
71
|
+
#: with ``B`` applied to ``A``'s arguments permuted by ``permutation`` — see
|
|
72
|
+
#: the module docstring for the exact axiom shape.
|
|
73
|
+
ConverseDeclaration = Tuple[_SymKey, _SymKey, Tuple[int, ...]]
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def validate_converses(declarations: Sequence[ConverseDeclaration]) -> None:
|
|
77
|
+
"""Raise :class:`ValueError` for any structurally invalid declaration.
|
|
78
|
+
|
|
79
|
+
Checked per declaration, in this order:
|
|
80
|
+
|
|
81
|
+
* ``a_key`` and ``b_key`` have the same arity (a converse cannot relate
|
|
82
|
+
predicates of different arity — there is no argument permutation
|
|
83
|
+
between them);
|
|
84
|
+
* ``permutation`` has exactly ``arity`` entries;
|
|
85
|
+
* ``permutation`` is a bijection on ``range(arity)`` (every position
|
|
86
|
+
0..arity-1 appears exactly once — a real permutation, not a partial or
|
|
87
|
+
repeating map that would silently drop or duplicate an argument);
|
|
88
|
+
* ``a_key != b_key`` (a predicate cannot be declared its own converse);
|
|
89
|
+
* neither predicate name is a built-in (``=``, ``≠``, ``<``, ``>``, ``≤``,
|
|
90
|
+
``≥`` — see :data:`unicode_logic_kit.eval.validate._BUILTIN_PREDS`), since
|
|
91
|
+
those already have fixed Z3 semantics an extra biconditional cannot
|
|
92
|
+
touch meaningfully and must not be allowed to appear to.
|
|
93
|
+
|
|
94
|
+
Checked across the whole list: the same UNORDERED pair ``{a_key, b_key}``
|
|
95
|
+
is never declared twice — this single check also catches two genuinely
|
|
96
|
+
CONTRADICTORY declarations for the same pair (e.g. two different
|
|
97
|
+
permutations for ``{LovedBy, Loves}``), since both shapes collide on the
|
|
98
|
+
same unordered-pair key. A chain of declarations over DISTINCT pairs
|
|
99
|
+
(``A~B``, ``B~C``) is explicitly allowed and composes correctly — each
|
|
100
|
+
axiom is independent, so the solver sees both biconditionals as separate
|
|
101
|
+
premises.
|
|
102
|
+
|
|
103
|
+
Never touches Z3 or any other backend — pure structural validation over
|
|
104
|
+
the declarations alone, so it is always safe (and cheap) to call before
|
|
105
|
+
committing to a solver run.
|
|
106
|
+
"""
|
|
107
|
+
seen_pairs: set = set()
|
|
108
|
+
for a_key, b_key, permutation in declarations:
|
|
109
|
+
a_name, a_arity = a_key
|
|
110
|
+
b_name, b_arity = b_key
|
|
111
|
+
|
|
112
|
+
if a_arity != b_arity:
|
|
113
|
+
raise ValueError(
|
|
114
|
+
f"validate_converses: arity mismatch between {a_key!r} and "
|
|
115
|
+
f"{b_key!r} — a converse permutation needs equal arity")
|
|
116
|
+
|
|
117
|
+
if len(permutation) != a_arity:
|
|
118
|
+
raise ValueError(
|
|
119
|
+
f"validate_converses: permutation {permutation!r} has "
|
|
120
|
+
f"{len(permutation)} entries, expected {a_arity} for {a_key!r}")
|
|
121
|
+
|
|
122
|
+
if sorted(permutation) != list(range(a_arity)):
|
|
123
|
+
raise ValueError(
|
|
124
|
+
f"validate_converses: permutation {permutation!r} for "
|
|
125
|
+
f"{a_key!r} is not a bijection on range({a_arity})")
|
|
126
|
+
|
|
127
|
+
if a_key == b_key:
|
|
128
|
+
raise ValueError(
|
|
129
|
+
f"validate_converses: {a_key!r} cannot be declared its own "
|
|
130
|
+
"converse")
|
|
131
|
+
|
|
132
|
+
for name in (a_name, b_name):
|
|
133
|
+
if name in _BUILTIN_PREDS:
|
|
134
|
+
raise ValueError(
|
|
135
|
+
f"validate_converses: {name!r} is a built-in predicate "
|
|
136
|
+
"and cannot be declared a converse")
|
|
137
|
+
|
|
138
|
+
pair = frozenset((a_key, b_key))
|
|
139
|
+
if pair in seen_pairs:
|
|
140
|
+
raise ValueError(
|
|
141
|
+
f"validate_converses: the pair {{{a_key!r}, {b_key!r}}} is "
|
|
142
|
+
"already declared — a predicate pair may only be declared "
|
|
143
|
+
"converses once (this also rejects two contradictory "
|
|
144
|
+
"permutations for the same pair)")
|
|
145
|
+
seen_pairs.add(pair)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def converse_axioms(declarations: Sequence[ConverseDeclaration]) -> Tuple[Node, ...]:
|
|
149
|
+
"""Build one closed biconditional sentence per declared converse.
|
|
150
|
+
|
|
151
|
+
Validates first (:func:`validate_converses` — every declaration must be
|
|
152
|
+
structurally sound before anything is built) and then, for each
|
|
153
|
+
``(a_key, b_key, permutation)`` with ``a_key = (a_name, arity)`` and
|
|
154
|
+
``b_key = (b_name, _)``, builds fresh bound variables ``v0 .. v_{n-1}``
|
|
155
|
+
(fresh PER AXIOM — reused across axioms is fine, since each axiom is its
|
|
156
|
+
own independently-scoped closed sentence, but never shared with the
|
|
157
|
+
caller's own formulas) and the sentence::
|
|
158
|
+
|
|
159
|
+
∀v0 … v_{n-1} (A(v0, …, v_{n-1}) ↔ B(v_perm[0], …, v_perm[n-1]))
|
|
160
|
+
|
|
161
|
+
i.e. ``A``'s arguments are the fresh variables in NATURAL order and
|
|
162
|
+
``B``'s are the same variables reordered by ``permutation`` — so
|
|
163
|
+
``permutation=(1, 0)`` for a binary pair yields
|
|
164
|
+
``A(v0, v1) ↔ B(v1, v0)``, the plain converse reading.
|
|
165
|
+
|
|
166
|
+
Each returned sentence is a pure biconditional between two otherwise-
|
|
167
|
+
unconstrained (uninterpreted) predicates — a DEFINITIONAL extension: for
|
|
168
|
+
any valuation of one side there is always a valuation of the other that
|
|
169
|
+
satisfies the axiom, so asserting it can never make a previously
|
|
170
|
+
satisfiable axiom set unsatisfiable, and can never manufacture an
|
|
171
|
+
entailment between predicates it does not mention (see
|
|
172
|
+
:mod:`unicode_logic_kit.eval.equivalence`'s module docstring for the
|
|
173
|
+
worked-through soundness argument this relies on).
|
|
174
|
+
|
|
175
|
+
Returns:
|
|
176
|
+
A tuple of closed :class:`~unicode_logic_kit.fol.nodes.Node` sentences,
|
|
177
|
+
same length and order as ``declarations``, each ``.to_z3()``-able and
|
|
178
|
+
``.to_unicode_str()``-able like any other formula node.
|
|
179
|
+
|
|
180
|
+
Raises:
|
|
181
|
+
ValueError: via :func:`validate_converses`, if any declaration is
|
|
182
|
+
structurally invalid.
|
|
183
|
+
"""
|
|
184
|
+
validate_converses(declarations)
|
|
185
|
+
|
|
186
|
+
axioms = []
|
|
187
|
+
for a_key, b_key, permutation in declarations:
|
|
188
|
+
a_name, arity = a_key
|
|
189
|
+
b_name, _b_arity = b_key
|
|
190
|
+
|
|
191
|
+
variables = tuple(Variable(f"v{j}") for j in range(arity))
|
|
192
|
+
atom_a = Atom(a_name, variables)
|
|
193
|
+
atom_b = Atom(b_name, tuple(variables[p] for p in permutation))
|
|
194
|
+
|
|
195
|
+
axiom: Node = Iff(atom_a, atom_b)
|
|
196
|
+
for v in reversed(variables):
|
|
197
|
+
axiom = Quantifier("∀", v, axiom)
|
|
198
|
+
axioms.append(axiom)
|
|
199
|
+
|
|
200
|
+
return tuple(axioms)
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
"""NL→logic benchmark adapters, plus a known_bad / self-audit mechanic for
|
|
2
|
+
finding defective gold FOL annotations.
|
|
3
|
+
|
|
4
|
+
Every adapter normalises its source rows into the SAME
|
|
5
|
+
:class:`DatasetExample` shape (see its docstring for the field mapping each
|
|
6
|
+
one uses), so downstream evaluation code can iterate a mixed pipeline without
|
|
7
|
+
caring which dataset an example came from:
|
|
8
|
+
|
|
9
|
+
from unicode_logic_kit.eval.datasets import load_folio, load_groves, audit_examples
|
|
10
|
+
|
|
11
|
+
examples = list(load_folio("folio-validation.jsonl", known_bad_ids={"folio:17"}))
|
|
12
|
+
report = audit_examples(examples) # finds parse/well-formedness defects
|
|
13
|
+
broken = [r for r in report if not r["ok"]]
|
|
14
|
+
|
|
15
|
+
The adapters (each module's docstring names its VERIFIED upstream source,
|
|
16
|
+
field schema, license, and every honest limitation; provenance is also
|
|
17
|
+
collected machine-readably in :data:`DATASET_INFO`):
|
|
18
|
+
|
|
19
|
+
- :mod:`~unicode_logic_kit.eval.datasets.folio` — FOLIO: premises/conclusion
|
|
20
|
+
with entailment labels AND gold FOL.
|
|
21
|
+
- :mod:`~unicode_logic_kit.eval.datasets.malls` — MALLS: NL→FOL translation
|
|
22
|
+
pairs (GPT-4-generated).
|
|
23
|
+
- :mod:`~unicode_logic_kit.eval.datasets.groves` — GROVES (this repo's author's
|
|
24
|
+
dataset): NL→FOL translation pairs in this kit's own Unicode notation.
|
|
25
|
+
- :mod:`~unicode_logic_kit.eval.datasets.willow` — WillowNLtoFOL: NL→FOL
|
|
26
|
+
translation pairs (one of GROVES's NL sources).
|
|
27
|
+
- :mod:`~unicode_logic_kit.eval.datasets.prontoqa` — ProntoQA (Logic-LM
|
|
28
|
+
rendering): NL QA plus a Logic-LM DSL program per example, with
|
|
29
|
+
:func:`~unicode_logic_kit.eval.datasets.prontoqa.parse_logic_program` (DSL →
|
|
30
|
+
kit AST) and :func:`~unicode_logic_kit.eval.datasets.prontoqa.solve_example`
|
|
31
|
+
(end-to-end deciding via ``api.prove``).
|
|
32
|
+
- :mod:`~unicode_logic_kit.eval.datasets.proofwriter` — ProofWriter, two
|
|
33
|
+
routes: :func:`load_proofwriter` reads the flat tasksource mirror (NL text
|
|
34
|
+
only, NO FOL gold; CWA/OWA label caveat in the module docstring), and
|
|
35
|
+
:func:`load_proofwriter_structured` reads the structured OWA distribution
|
|
36
|
+
and GENERATES FOL deterministically from the dataset's own triple/rule
|
|
37
|
+
representations (marked ``meta["fol_generated"]``;
|
|
38
|
+
:func:`~unicode_logic_kit.eval.datasets.proofwriter.solve_structured_example`
|
|
39
|
+
decides each question against its OWA label).
|
|
40
|
+
- :mod:`~unicode_logic_kit.eval.datasets.logicnli` — LogicNLI: NLI-style FOL
|
|
41
|
+
reasoning; the upstream structured logic annotation is preserved in
|
|
42
|
+
``meta``, not presented as gold FOL formula strings (it is not).
|
|
43
|
+
- :mod:`~unicode_logic_kit.eval.datasets.proverqa` — ProverQA: FOL-annotated QA
|
|
44
|
+
whose gold formulas use the OPPOSITE naming convention from this kit's
|
|
45
|
+
grammar (snake_case predicates, capitalised constants). The loader parses
|
|
46
|
+
them with a dedicated import-time dialect grammar and re-emits kit
|
|
47
|
+
notation (originals + injective name mapping in ``meta``;
|
|
48
|
+
``convert_fol=False`` restores the verbatim 0%-parse pass-through);
|
|
49
|
+
:func:`~unicode_logic_kit.eval.datasets.proverqa.solve_example` decides each
|
|
50
|
+
example end-to-end. Willow gets the same treatment for its ~1.2% tail via
|
|
51
|
+
:func:`~unicode_logic_kit.eval.datasets.willow.repair_willow_formula`
|
|
52
|
+
(NLTK-precedence reading, minimal renaming).
|
|
53
|
+
|
|
54
|
+
- :mod:`~unicode_logic_kit.eval.datasets.fracas` — FraCaS: the one PURE-NLI
|
|
55
|
+
adapter, premises/hypothesis/three-valued answer with no logic annotation
|
|
56
|
+
anywhere (its gold ``yes``/``no``/``unknown`` maps onto ``⊨ h`` / ``⊨ ¬h``
|
|
57
|
+
/ neither exactly, so it is a reference target for a translation step that
|
|
58
|
+
lives outside this library —
|
|
59
|
+
:func:`~unicode_logic_kit.eval.datasets.fracas.solve_example` takes that
|
|
60
|
+
translation as an injected callable). ``fol_premises`` is always empty, so
|
|
61
|
+
:func:`audit_examples` is vacuous on it;
|
|
62
|
+
:func:`~unicode_logic_kit.eval.datasets.fracas.ace_census` reports, per
|
|
63
|
+
sentence, what APE accepts as controlled English.
|
|
64
|
+
- :mod:`~unicode_logic_kit.eval.datasets.pmb` — PMB (Parallel Meaning Bank): the
|
|
65
|
+
only adapter with no ready-made NL/FOL pair file — gold FOL is produced by
|
|
66
|
+
reading each document's own DRS (in SBN notation) through
|
|
67
|
+
:mod:`unicode_logic_kit.drt`, and a document this subset's grammar cannot
|
|
68
|
+
parse still yields an example, with the refusal recorded in
|
|
69
|
+
``meta["parse_error"]`` rather than dropped.
|
|
70
|
+
- :mod:`~unicode_logic_kit.eval.datasets.pfolio` — P-FOLIO: FOLIO's own
|
|
71
|
+
entailment labels, joined against a bundled, human-written step-by-step
|
|
72
|
+
derivation per conclusion (``meta["proof_steps"]``); every join is
|
|
73
|
+
cross-checked against ``FOLIO.csv``'s own truth value, never merged
|
|
74
|
+
silently, and a disagreement is refused rather than guessed
|
|
75
|
+
(:func:`~unicode_logic_kit.eval.datasets.pfolio.pfolio_refusals`).
|
|
76
|
+
- :mod:`~unicode_logic_kit.eval.datasets.logicbench` — LogicBench: 25
|
|
77
|
+
single-inference-rule reasoning patterns split by logic type
|
|
78
|
+
(propositional, first-order, non-monotonic), another no-gold-FOL adapter in
|
|
79
|
+
FraCaS's shape;
|
|
80
|
+
:func:`~unicode_logic_kit.eval.datasets.logicbench.solve_example` routes a
|
|
81
|
+
non-monotonic-logic row through :mod:`unicode_logic_kit.semantics.nonmonotonic`
|
|
82
|
+
instead of a second classical cascade.
|
|
83
|
+
|
|
84
|
+
Per-dataset helper functions (converters, solvers) deliberately live on
|
|
85
|
+
their OWN modules rather than being re-exported here — two adapters
|
|
86
|
+
legitimately name their solver ``solve_example``, and a package-level
|
|
87
|
+
re-export would force one to shadow the other.
|
|
88
|
+
|
|
89
|
+
Deliberately NOT adapted (verified 2026-08-12, so the omission is a decision,
|
|
90
|
+
not an oversight): **AR-LSAT** (no logic annotation of any kind — Logic-LM
|
|
91
|
+
generates a Z3-style DSL at inference time, which is neither gold data nor
|
|
92
|
+
FOL; and the underlying LSAT passages are copyright-encumbered),
|
|
93
|
+
and **LogicalDeduction** (BIG-bench; pure NL ordering puzzles, no formal
|
|
94
|
+
annotation — Logic-LM uses a runtime CSP DSL). FraCaS was listed here for
|
|
95
|
+
the same reason until it got the pure-NLI adapter above; its source file
|
|
96
|
+
carries no explicit licence statement, so this repository ships a synthetic
|
|
97
|
+
fixture and reads the real file from a caller-supplied path.
|
|
98
|
+
|
|
99
|
+
No loader downloads anything — every one reads a LOCAL file the caller
|
|
100
|
+
already obtained (some upstreams distribute JSON arrays; each adapter's
|
|
101
|
+
docstring gives the one-line conversion recipe).
|
|
102
|
+
|
|
103
|
+
``known_bad`` (set by the loader from a caller-supplied ``known_bad_ids``) is
|
|
104
|
+
a CURATED flag; :func:`audit_examples` is an AUTOMATED, independent check
|
|
105
|
+
(does every gold FOL string parse, and is it well-formed?) — the two are
|
|
106
|
+
meant to be cross-checked against each other, not conflated as the same
|
|
107
|
+
signal.
|
|
108
|
+
"""
|
|
109
|
+
|
|
110
|
+
from ._base import DatasetExample, DATASET_INFO, audit_examples
|
|
111
|
+
from .folio import load_folio
|
|
112
|
+
from .malls import load_malls
|
|
113
|
+
from .groves import load_groves
|
|
114
|
+
from .willow import load_willow
|
|
115
|
+
from .prontoqa import load_prontoqa
|
|
116
|
+
from .proofwriter import load_proofwriter, load_proofwriter_structured
|
|
117
|
+
from .logicnli import load_logicnli
|
|
118
|
+
from .proverqa import load_proverqa
|
|
119
|
+
from .fracas import load_fracas
|
|
120
|
+
# C3PO is the first adapter whose gold is not a formula but an EXECUTABLE
|
|
121
|
+
# membership decision: a definition is scored by model-checking it against
|
|
122
|
+
# real molecule structures, so score_definition lives here beside the loader.
|
|
123
|
+
from .c3po import load_c3po, score_definition, DefinitionScore
|
|
124
|
+
from .pmb import load_pmb
|
|
125
|
+
from .pfolio import load_pfolio
|
|
126
|
+
from .logicbench import load_logicbench
|
|
127
|
+
|
|
128
|
+
__all__ = [
|
|
129
|
+
"DatasetExample", "DATASET_INFO", "audit_examples",
|
|
130
|
+
"load_folio", "load_malls", "load_groves", "load_willow",
|
|
131
|
+
"load_prontoqa",
|
|
132
|
+
"load_proofwriter", "load_proofwriter_structured",
|
|
133
|
+
"load_logicnli", "load_proverqa", "load_fracas",
|
|
134
|
+
"load_c3po", "score_definition", "DefinitionScore",
|
|
135
|
+
"load_pmb", "load_pfolio", "load_logicbench",
|
|
136
|
+
]
|
|
@@ -0,0 +1,263 @@
|
|
|
1
|
+
"""Shared infrastructure for NL/FOL dataset adapters (:mod:`folio`, :mod:`malls`).
|
|
2
|
+
|
|
3
|
+
This module defines the one example shape every dataset adapter normalises
|
|
4
|
+
into (:class:`DatasetExample`), a small provenance registry
|
|
5
|
+
(:data:`DATASET_INFO`), and the self-audit mechanic (:func:`audit_examples`)
|
|
6
|
+
that a caller uses to find defective *gold* formulas in a dataset BEFORE
|
|
7
|
+
trusting them as an evaluation reference — an LLM-authored or crowd-annotated
|
|
8
|
+
FOL string can itself be unparseable or ill-formed, and treating it as ground
|
|
9
|
+
truth without checking would silently corrupt every metric computed against
|
|
10
|
+
it.
|
|
11
|
+
|
|
12
|
+
Design notes:
|
|
13
|
+
|
|
14
|
+
* :class:`DatasetExample` is deliberately dataset-agnostic: FOLIO's
|
|
15
|
+
premises/conclusion/label entailment shape and MALLS's single
|
|
16
|
+
NL-statement/FOL-formula translation pairs both fit the same fields (see
|
|
17
|
+
each adapter module's docstring for its exact field mapping).
|
|
18
|
+
* Parsing is LAZY and never raises: :meth:`DatasetExample.parse_premises` /
|
|
19
|
+
:meth:`DatasetExample.parse_conclusion` call
|
|
20
|
+
:func:`unicode_logic_kit.api.parse_any` only when invoked (loading a dataset
|
|
21
|
+
of thousands of examples must not eagerly parse every one), and return
|
|
22
|
+
:class:`~unicode_logic_kit.api.ParseResult` objects — a parse failure is
|
|
23
|
+
reported, never swallowed or turned into an exception.
|
|
24
|
+
* ``known_bad`` is a CURATED flag an adapter sets from a caller-supplied
|
|
25
|
+
``known_bad_ids`` set (e.g. IDs a human or a prior audit run identified as
|
|
26
|
+
having a broken gold annotation). It is independent of
|
|
27
|
+
:func:`audit_examples`, which is the AUTOMATED detector: the two are meant
|
|
28
|
+
to be cross-checked against each other, not conflated.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from dataclasses import dataclass, field
|
|
32
|
+
from typing import TYPE_CHECKING, Dict, Iterable, List, Optional, Tuple
|
|
33
|
+
|
|
34
|
+
if TYPE_CHECKING: # pragma: no cover - typing only
|
|
35
|
+
from ...api import CheckResult, ParseResult
|
|
36
|
+
|
|
37
|
+
__all__ = ["DatasetExample", "DATASET_INFO", "audit_examples"]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
# ---------------------------------------------------------------------------
|
|
41
|
+
# DatasetExample
|
|
42
|
+
# ---------------------------------------------------------------------------
|
|
43
|
+
|
|
44
|
+
@dataclass(frozen=True)
|
|
45
|
+
class DatasetExample:
|
|
46
|
+
"""One normalised NL/FOL example from any adapter in this subpackage.
|
|
47
|
+
|
|
48
|
+
Fields:
|
|
49
|
+
id: a stable identifier. Adapters use the dataset's own ID field when
|
|
50
|
+
the source data provides one, otherwise a positional fallback
|
|
51
|
+
(documented per adapter) — never ``None``, so examples are always
|
|
52
|
+
addressable (e.g. for ``known_bad_ids``).
|
|
53
|
+
nl_premises: natural-language premise sentences, in source order.
|
|
54
|
+
Empty for datasets with no premise/conclusion structure (e.g.
|
|
55
|
+
MALLS — see its adapter module for the field-mapping rationale).
|
|
56
|
+
fol_premises: the gold FOL premise strings, in the SAME order as
|
|
57
|
+
``nl_premises`` (index ``i`` of one corresponds to index ``i`` of
|
|
58
|
+
the other). Left as raw strings, not parsed, until
|
|
59
|
+
:meth:`parse_premises` is called.
|
|
60
|
+
nl_conclusion / fol_conclusion: the natural-language / FOL conclusion
|
|
61
|
+
(entailment datasets), or the single NL statement / FOL formula
|
|
62
|
+
being translated (translation-pair datasets). ``None`` when the
|
|
63
|
+
dataset has no conclusion slot at all.
|
|
64
|
+
label: the gold entailment label (dataset-specific vocabulary, e.g.
|
|
65
|
+
FOLIO's ``"True"``/``"False"``/``"Uncertain"``), or ``None`` for
|
|
66
|
+
datasets without one (e.g. MALLS).
|
|
67
|
+
known_bad: True iff this example's ``id`` was passed in the loader's
|
|
68
|
+
``known_bad_ids`` — a caller-curated "do not trust this gold
|
|
69
|
+
annotation" flag, set at load time and never inferred here.
|
|
70
|
+
meta: adapter-specific leftover fields from the source record (e.g.
|
|
71
|
+
FOLIO's ``source``/``story-id``, or a synthetic-fixture marker),
|
|
72
|
+
kept as a plain JSON-compatible dict so nothing from the original
|
|
73
|
+
record is silently discarded.
|
|
74
|
+
"""
|
|
75
|
+
|
|
76
|
+
id: str
|
|
77
|
+
nl_premises: Tuple[str, ...]
|
|
78
|
+
fol_premises: Tuple[str, ...]
|
|
79
|
+
nl_conclusion: Optional[str]
|
|
80
|
+
fol_conclusion: Optional[str]
|
|
81
|
+
label: Optional[str]
|
|
82
|
+
known_bad: bool
|
|
83
|
+
meta: dict = field(default_factory=dict)
|
|
84
|
+
|
|
85
|
+
def parse_premises(self) -> Tuple["ParseResult", ...]:
|
|
86
|
+
"""Parse every ``fol_premises`` string, lazily, in order.
|
|
87
|
+
|
|
88
|
+
Uses :func:`unicode_logic_kit.api.parse_any` (dialect auto-detection);
|
|
89
|
+
each element of the returned tuple is that call's
|
|
90
|
+
:class:`~unicode_logic_kit.api.ParseResult` UNCHANGED — a parse failure
|
|
91
|
+
shows up as ``ok=False`` with its ``errors``, never as an exception
|
|
92
|
+
and never silently dropped.
|
|
93
|
+
"""
|
|
94
|
+
from ... import api # lazy: avoid import-time cost/cycles
|
|
95
|
+
|
|
96
|
+
return tuple(api.parse_any(p) for p in self.fol_premises)
|
|
97
|
+
|
|
98
|
+
def parse_conclusion(self) -> Optional["ParseResult"]:
|
|
99
|
+
"""Parse ``fol_conclusion``, lazily — ``None`` iff there is none.
|
|
100
|
+
|
|
101
|
+
Returns the raw :class:`~unicode_logic_kit.api.ParseResult` (so a parse
|
|
102
|
+
failure is visible via ``ok=False``/``errors``), or ``None`` when this
|
|
103
|
+
example carries no conclusion/target formula at all (a different
|
|
104
|
+
thing from a formula that failed to parse).
|
|
105
|
+
"""
|
|
106
|
+
if self.fol_conclusion is None:
|
|
107
|
+
return None
|
|
108
|
+
from ... import api # lazy: avoid import-time cost/cycles
|
|
109
|
+
|
|
110
|
+
return api.parse_any(self.fol_conclusion)
|
|
111
|
+
|
|
112
|
+
def to_dict(self) -> dict:
|
|
113
|
+
"""Serialise to a JSON-compatible dict (all fields, names as keys)."""
|
|
114
|
+
return {
|
|
115
|
+
"id": self.id,
|
|
116
|
+
"nl_premises": list(self.nl_premises),
|
|
117
|
+
"fol_premises": list(self.fol_premises),
|
|
118
|
+
"nl_conclusion": self.nl_conclusion,
|
|
119
|
+
"fol_conclusion": self.fol_conclusion,
|
|
120
|
+
"label": self.label,
|
|
121
|
+
"known_bad": self.known_bad,
|
|
122
|
+
"meta": self.meta,
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
# ---------------------------------------------------------------------------
|
|
127
|
+
# Provenance registry
|
|
128
|
+
# ---------------------------------------------------------------------------
|
|
129
|
+
|
|
130
|
+
#: Dataset name -> provenance dict (``license``, ``source_url``,
|
|
131
|
+
#: ``citation_hint``). Populated by each adapter module at import time via
|
|
132
|
+
#: :func:`_register_dataset_info` so this dict never drifts from the loader
|
|
133
|
+
#: that actually claims to implement it. Verified against the primary sources
|
|
134
|
+
#: named in each adapter's module docstring — nothing here is guessed.
|
|
135
|
+
DATASET_INFO: Dict[str, dict] = {}
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _register_dataset_info(name: str, *, license: str, source_url: str,
|
|
139
|
+
citation_hint: str) -> None:
|
|
140
|
+
"""Record ``name``'s provenance in :data:`DATASET_INFO` (adapter-internal)."""
|
|
141
|
+
DATASET_INFO[name] = {
|
|
142
|
+
"license": license,
|
|
143
|
+
"source_url": source_url,
|
|
144
|
+
"citation_hint": citation_hint,
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
# ---------------------------------------------------------------------------
|
|
149
|
+
# audit_examples — the self-audit mechanic
|
|
150
|
+
# ---------------------------------------------------------------------------
|
|
151
|
+
|
|
152
|
+
def _check_result_defects(check_result: "CheckResult", field_name: str,
|
|
153
|
+
index: Optional[int]) -> List[dict]:
|
|
154
|
+
"""Expand one failing :class:`~unicode_logic_kit.api.CheckResult` into
|
|
155
|
+
per-aspect defect dicts (a formula can fail more than one aspect at once,
|
|
156
|
+
e.g. an unbound variable AND an arity conflict, so this can return more
|
|
157
|
+
than one entry for a single formula).
|
|
158
|
+
"""
|
|
159
|
+
defects: List[dict] = []
|
|
160
|
+
if not check_result.is_closed:
|
|
161
|
+
defects.append({
|
|
162
|
+
"kind": "free_variables", "field": field_name, "index": index,
|
|
163
|
+
"detail": list(check_result.free_variables),
|
|
164
|
+
})
|
|
165
|
+
if not check_result.arity_consistent:
|
|
166
|
+
defects.append({
|
|
167
|
+
"kind": "arity_conflict", "field": field_name, "index": index,
|
|
168
|
+
"detail": list(check_result.arity_conflicts),
|
|
169
|
+
})
|
|
170
|
+
if check_result.has_lambdas:
|
|
171
|
+
defects.append({
|
|
172
|
+
"kind": "residual_lambda", "field": field_name, "index": index,
|
|
173
|
+
"detail": None,
|
|
174
|
+
})
|
|
175
|
+
return defects
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def audit_examples(examples: Iterable[DatasetExample], *,
|
|
179
|
+
max_examples: Optional[int] = None) -> List[dict]:
|
|
180
|
+
"""Self-audit a batch of examples: which gold FOL strings are broken?
|
|
181
|
+
|
|
182
|
+
This is the mechanic for finding defective gold annotations WITHOUT any
|
|
183
|
+
external reference — it only requires that a gold formula (a) parses and
|
|
184
|
+
(b) is well-formed (closed, arity-consistent, lambda-free), both of which
|
|
185
|
+
:func:`unicode_logic_kit.api.parse_any` / :func:`unicode_logic_kit.api.check`
|
|
186
|
+
can decide on their own. It does NOT check the entailment ``label``
|
|
187
|
+
itself (that would need a prover run over the whole premise set, a
|
|
188
|
+
heavier and separately-scoped operation).
|
|
189
|
+
|
|
190
|
+
Args:
|
|
191
|
+
examples: any iterable of :class:`DatasetExample` (a generator from a
|
|
192
|
+
loader is fine — nothing here requires a materialised list).
|
|
193
|
+
max_examples: stop after auditing this many examples (``None`` audits
|
|
194
|
+
everything). Useful for a quick pass over a large dataset.
|
|
195
|
+
|
|
196
|
+
Returns:
|
|
197
|
+
A ``list[dict]``, one entry per audited example, in input order, each
|
|
198
|
+
JSON-compatible:
|
|
199
|
+
|
|
200
|
+
* ``id`` / ``known_bad``: copied from the example, for cross-checking
|
|
201
|
+
the automated finding against any curated flag.
|
|
202
|
+
* ``all_parsed``: True iff every ``fol_premises`` string and (if
|
|
203
|
+
present) ``fol_conclusion`` parsed successfully — answers "parsen
|
|
204
|
+
alle FOL-Strings?".
|
|
205
|
+
* ``all_checked``: True iff every formula that DID parse also passed
|
|
206
|
+
:func:`~unicode_logic_kit.api.check` (``check().ok``); vacuously True
|
|
207
|
+
if nothing parsed. Independent of ``all_parsed`` — a formula that
|
|
208
|
+
fails to parse contributes nothing here, only to ``all_parsed``.
|
|
209
|
+
* ``ok``: ``all_parsed and all_checked`` — no defect of any kind.
|
|
210
|
+
* ``defects``: the collected defect classes, each
|
|
211
|
+
``{"kind": ..., "field": "premises"|"conclusion", "index": int|None,
|
|
212
|
+
"detail": ...}``. ``kind`` is one of ``"unparseable"``,
|
|
213
|
+
``"free_variables"``, ``"arity_conflict"``, ``"residual_lambda"``.
|
|
214
|
+
``index`` is the position within ``fol_premises`` for that field, or
|
|
215
|
+
``None`` for the (singular) conclusion.
|
|
216
|
+
"""
|
|
217
|
+
from ... import api # lazy: avoid import-time cost/cycles
|
|
218
|
+
|
|
219
|
+
reports: List[dict] = []
|
|
220
|
+
for i, example in enumerate(examples):
|
|
221
|
+
if max_examples is not None and i >= max_examples:
|
|
222
|
+
break
|
|
223
|
+
|
|
224
|
+
defects: List[dict] = []
|
|
225
|
+
all_parsed = True
|
|
226
|
+
all_checked = True
|
|
227
|
+
|
|
228
|
+
premise_results = example.parse_premises()
|
|
229
|
+
for idx, pr in enumerate(premise_results):
|
|
230
|
+
if not pr.ok:
|
|
231
|
+
all_parsed = False
|
|
232
|
+
message = pr.errors[-1]["message"] if pr.errors else "unparseable"
|
|
233
|
+
defects.append({"kind": "unparseable", "field": "premises",
|
|
234
|
+
"index": idx, "detail": message})
|
|
235
|
+
continue
|
|
236
|
+
cr = api.check(pr.formula)
|
|
237
|
+
if not cr.ok:
|
|
238
|
+
all_checked = False
|
|
239
|
+
defects.extend(_check_result_defects(cr, "premises", idx))
|
|
240
|
+
|
|
241
|
+
conclusion_result = example.parse_conclusion()
|
|
242
|
+
if conclusion_result is not None:
|
|
243
|
+
if not conclusion_result.ok:
|
|
244
|
+
all_parsed = False
|
|
245
|
+
message = (conclusion_result.errors[-1]["message"]
|
|
246
|
+
if conclusion_result.errors else "unparseable")
|
|
247
|
+
defects.append({"kind": "unparseable", "field": "conclusion",
|
|
248
|
+
"index": None, "detail": message})
|
|
249
|
+
else:
|
|
250
|
+
cr = api.check(conclusion_result.formula)
|
|
251
|
+
if not cr.ok:
|
|
252
|
+
all_checked = False
|
|
253
|
+
defects.extend(_check_result_defects(cr, "conclusion", None))
|
|
254
|
+
|
|
255
|
+
reports.append({
|
|
256
|
+
"id": example.id,
|
|
257
|
+
"known_bad": example.known_bad,
|
|
258
|
+
"all_parsed": all_parsed,
|
|
259
|
+
"all_checked": all_checked,
|
|
260
|
+
"ok": all_parsed and all_checked,
|
|
261
|
+
"defects": defects,
|
|
262
|
+
})
|
|
263
|
+
return reports
|