unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,466 @@
|
|
|
1
|
+
"""Graded equivalence checking for NL→FOL evaluation.
|
|
2
|
+
|
|
3
|
+
``equivalent(prediction, reference)`` is the one entry point that answers "does
|
|
4
|
+
the predicted formula mean the same as the reference?" at every level the kit
|
|
5
|
+
can decide, from cheap structural checks to a solver call:
|
|
6
|
+
|
|
7
|
+
* ``exact`` — structural equality of the frozen ASTs (``==``);
|
|
8
|
+
* ``canonical`` — :func:`~unicode_logic_kit.eval.canonical.exact_match`
|
|
9
|
+
(α-renaming, commutativity/associativity, operand
|
|
10
|
+
de-duplication, double negation quotiented out);
|
|
11
|
+
* ``predicate_align`` — :func:`~unicode_logic_kit.eval.predicate_match.aligned_exact_match`
|
|
12
|
+
(vocabulary differences quotiented out too:
|
|
13
|
+
namespace- and arity-aware injective renaming,
|
|
14
|
+
then canonical match);
|
|
15
|
+
* ``solver`` — genuine logical equivalence. Classical formulas go
|
|
16
|
+
to Z3 with a TRI-STATE result (``True`` proved /
|
|
17
|
+
``False`` refuted with a counterexample /
|
|
18
|
+
``None`` unknown — the existing
|
|
19
|
+
``formulas_are_equivalent`` collapses unknown to
|
|
20
|
+
False, which a metric must not do). Modal formulas
|
|
21
|
+
go to ``modal_decide`` (tri-state, with a Kripke
|
|
22
|
+
counterexample on refutation) and fall back to the
|
|
23
|
+
sound-but-bounded-incomplete ``qml_equivalent``
|
|
24
|
+
(``True`` / ``None``) for fragments the tableau
|
|
25
|
+
rejects;
|
|
26
|
+
* ``auto`` — the ladder above, cheapest first, stopping at the
|
|
27
|
+
first level that answers ``True``.
|
|
28
|
+
|
|
29
|
+
An OPT-IN ``converses`` argument additionally lets a caller declare
|
|
30
|
+
argument-permutation bridging axioms — e.g. ``LovedBy(x, y) ↔ Loves(y, x)`` —
|
|
31
|
+
that the SOLVER level asserts as extra premises (see
|
|
32
|
+
:mod:`unicode_logic_kit.eval.converses`). Only ``method="solver"``/``"auto"``
|
|
33
|
+
ever honour it (a structural level would silently ignore a declared axiom it
|
|
34
|
+
cannot consume, which is refused rather than allowed); the result is tagged
|
|
35
|
+
with its own ``method_used`` value, ``"solver_modulo_converses"``, never
|
|
36
|
+
merged into plain ``"solver"`` so a consumer can always tell whether a
|
|
37
|
+
verdict relied on caller-declared bridges. This kit's whole classical Z3
|
|
38
|
+
export uses exactly one Z3 sort for every term (see
|
|
39
|
+
``eval.converses``'s module docstring), so a declared axiom for a
|
|
40
|
+
``(name, arity)`` predicate pair interns to the identical Z3 function the
|
|
41
|
+
compared formulas themselves use — many-sorted (``SortedQuantifier``) input
|
|
42
|
+
included; ``tests/test_converses.py`` proves this end-to-end rather than just
|
|
43
|
+
asserting it.
|
|
44
|
+
|
|
45
|
+
Every result is an :class:`EquivalenceResult` whose ``to_dict()`` is
|
|
46
|
+
JSON-compatible, so pipelines and LLM repair loops can consume it without
|
|
47
|
+
extra parsing.
|
|
48
|
+
|
|
49
|
+
``partial_credit`` (``method="auto"`` / ``"solver"`` only) is a heuristic
|
|
50
|
+
RANKING signal for use when the headline verdict is not a clean ``True`` —
|
|
51
|
+
NOT a probability, just a coarse {0, 0.25, 0.5, 0.75, 1.0} ordinal built from
|
|
52
|
+
four binary sub-checks (well-formedness, vocabulary alignability, aligned
|
|
53
|
+
structural match, logical equivalence). See :class:`EquivalenceResult` for
|
|
54
|
+
the exact rubric.
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
from dataclasses import dataclass
|
|
58
|
+
from typing import Optional
|
|
59
|
+
|
|
60
|
+
from unicode_logic_kit.fol._msfl_nodes import sort_axioms
|
|
61
|
+
from unicode_logic_kit.fol.nodes import (
|
|
62
|
+
Node, Iff, Quantifier, SortedQuantifier,
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
__all__ = ["EquivalenceResult", "equivalent"]
|
|
66
|
+
|
|
67
|
+
_METHODS = ("exact", "canonical", "predicate_align", "solver", "auto")
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
@dataclass(frozen=True)
|
|
71
|
+
class EquivalenceResult:
|
|
72
|
+
"""Outcome of :func:`equivalent` — one Optional[bool] per level.
|
|
73
|
+
|
|
74
|
+
``None`` always means "not computed" for the structural levels, and
|
|
75
|
+
"could not decide within budget" for ``logically_equivalent``.
|
|
76
|
+
|
|
77
|
+
Fields:
|
|
78
|
+
|
|
79
|
+
``equivalent``
|
|
80
|
+
the headline verdict — ``True`` iff the requested method (or, for
|
|
81
|
+
``auto``, any level of the ladder) established equivalence; ``False``
|
|
82
|
+
iff the strongest level that ran refuted it; ``None`` if undecided.
|
|
83
|
+
Truthiness follows this field.
|
|
84
|
+
``method_used``
|
|
85
|
+
the level that produced the headline verdict. Almost always one of
|
|
86
|
+
``"exact"`` / ``"canonical"`` / ``"predicate_align"`` / ``"solver"``
|
|
87
|
+
(matching the ``method`` argument's own vocabulary); the one
|
|
88
|
+
exception is ``"solver_modulo_converses"`` — set instead of
|
|
89
|
+
``"solver"`` exactly when the caller passed a non-empty
|
|
90
|
+
``converses`` AND the solver level ran, regardless of the tri-state
|
|
91
|
+
outcome, so a consumer can always tell a plain solver verdict from
|
|
92
|
+
one that relied on caller-declared converse axioms (see
|
|
93
|
+
:func:`equivalent`'s ``converses`` parameter and
|
|
94
|
+
:mod:`unicode_logic_kit.eval.converses`).
|
|
95
|
+
``syntax_equal`` / ``structurally_equal`` / ``aligned_equal``
|
|
96
|
+
the three rungs of the structural ladder: ``==``, canonical form,
|
|
97
|
+
and alignment followed by canonical form.
|
|
98
|
+
``logically_equivalent``
|
|
99
|
+
the solver's tri-state verdict.
|
|
100
|
+
``counterexample``
|
|
101
|
+
THE COUNTERMODEL, as an accessible, documented, JSON-able field —
|
|
102
|
+
set exactly when the solver stage REFUTED equivalence (i.e. whenever
|
|
103
|
+
``equivalent is False``; for ``method="solver"``/``"auto"`` that is
|
|
104
|
+
the same moment ``logically_equivalent is False``), ``None`` in
|
|
105
|
+
every other case (proved, undecided, or a structural level never
|
|
106
|
+
reached the solver). A caller checks it with
|
|
107
|
+
``if result.counterexample is not None: ...`` — no separate lookup
|
|
108
|
+
or solver re-invocation needed; the witness that falsified
|
|
109
|
+
equivalence is already attached to the result that reports the
|
|
110
|
+
refutation. Two shapes, keyed by ``"kind"``:
|
|
111
|
+
``{"kind": "z3_model", "assignment": {var_name: value_repr, …}}``
|
|
112
|
+
for classical formulas (every declared Z3 symbol's value under the
|
|
113
|
+
model that satisfies ``¬(φ ↔ ψ)``), and
|
|
114
|
+
``{"kind": "kripke", "repr": …}`` for modal ones (the ``repr()`` of
|
|
115
|
+
the Kripke countermodel :func:`~unicode_logic_kit.atp.modal_tableau.modal_countermodel`
|
|
116
|
+
found). Both are plain JSON-compatible dicts of strings, so the
|
|
117
|
+
whole ``EquivalenceResult`` — countermodel included — survives a
|
|
118
|
+
``pickle`` round-trip and a process boundary undisturbed (it is a
|
|
119
|
+
frozen dataclass over ``Optional[bool]``/``str``/``dict`` fields
|
|
120
|
+
only), which matters for a caller that evaluates candidates in a
|
|
121
|
+
worker pool and inspects failures back in the parent process.
|
|
122
|
+
``reason``
|
|
123
|
+
why the solver level gave no verdict, when it REFUSED the input: the
|
|
124
|
+
refusal's own text (a family with no first-order image, a numeral and a
|
|
125
|
+
constant that are one symbol, a construct the modal route does not
|
|
126
|
+
decide), so that ``equivalent is None`` can be told from "undecided within
|
|
127
|
+
the budget". ``None`` in every other case: a verdict, an undecided search,
|
|
128
|
+
or a call that never reached the solver.
|
|
129
|
+
``partial_credit``
|
|
130
|
+
a heuristic RANKING signal in ``{0.0, 0.25, 0.5, 0.75, 1.0}`` —
|
|
131
|
+
explicitly NOT a probability of correctness, just a coarse ordinal
|
|
132
|
+
score for sorting or averaging predictions when the headline verdict
|
|
133
|
+
is not a clean ``True``. The rubric: ``equivalent is True`` at ANY
|
|
134
|
+
level of the ladder scores ``1.0`` (the headline already gives the
|
|
135
|
+
strongest possible signal, so the sub-component breakdown is not
|
|
136
|
+
computed and ``partial_credit_components`` stays ``None``); otherwise
|
|
137
|
+
the score is the sum of four independent 0.25-weighted binary checks,
|
|
138
|
+
see ``partial_credit_components``.
|
|
139
|
+
|
|
140
|
+
Only computed for ``method="auto"`` and ``method="solver"``: the
|
|
141
|
+
rubric's ``s4`` component needs the solver's tri-state verdict, which
|
|
142
|
+
the purely structural methods (``exact`` / ``canonical`` /
|
|
143
|
+
``predicate_align``) never obtain — for those, ``partial_credit``
|
|
144
|
+
stays ``None`` (an honest "not computed", never a score of 0).
|
|
145
|
+
``partial_credit_components``
|
|
146
|
+
the four binary sub-checks behind a summed (non-``1.0``,
|
|
147
|
+
non-``None``) ``partial_credit``, as a JSON-able
|
|
148
|
+
``{"s1": bool, "s2": bool, "s3": bool, "s4": bool}``. ``None``
|
|
149
|
+
whenever ``partial_credit`` itself is ``None`` or ``1.0`` (see
|
|
150
|
+
above). The four keys:
|
|
151
|
+
|
|
152
|
+
* ``s1`` "wellformed" — BOTH prediction and reference are
|
|
153
|
+
``eval.validate.is_wellformed`` (closed, arity-consistent,
|
|
154
|
+
lambda-free).
|
|
155
|
+
* ``s2`` "vocabulary_alignable" — after
|
|
156
|
+
``align_symbols(prediction, reference)`` the aligned prediction's
|
|
157
|
+
symbol inventories (predicate name/arity, function name/arity,
|
|
158
|
+
constant names) equal the reference's.
|
|
159
|
+
* ``s3`` "aligned_equal" — ``aligned_exact_match`` holds of the pair.
|
|
160
|
+
This practically implies ``s2`` (a successful vocabulary alignment
|
|
161
|
+
is a prerequisite for the canonical match that follows it) but is
|
|
162
|
+
tracked as an independent bit because it is the STRICTLY stronger
|
|
163
|
+
condition — matching vocabularies is necessary but not sufficient
|
|
164
|
+
for the aligned canonical forms to agree.
|
|
165
|
+
* ``s4`` "logically_equivalent" — the solver's tri-state verdict is
|
|
166
|
+
True; both ``None`` (undecided) and ``False`` (refuted) count as 0,
|
|
167
|
+
per the tri-state discipline the rest of this module enforces (see
|
|
168
|
+
the module docstring). Note: for the two methods that ever reach
|
|
169
|
+
the summed branch, the headline verdict IS the solver verdict, so
|
|
170
|
+
by the time ``s1``–``s4`` are computed that verdict has already
|
|
171
|
+
failed to be ``True`` — meaning ``s4`` is always ``False`` in every
|
|
172
|
+
case this field is actually populated today. It is still evaluated
|
|
173
|
+
against the real verdict (never hard-coded) so it stays correct
|
|
174
|
+
should a future call site ever decouple the two.
|
|
175
|
+
"""
|
|
176
|
+
|
|
177
|
+
equivalent: Optional[bool]
|
|
178
|
+
method_used: str
|
|
179
|
+
syntax_equal: Optional[bool] = None
|
|
180
|
+
structurally_equal: Optional[bool] = None
|
|
181
|
+
aligned_equal: Optional[bool] = None
|
|
182
|
+
logically_equivalent: Optional[bool] = None
|
|
183
|
+
counterexample: Optional[dict] = None
|
|
184
|
+
partial_credit: Optional[float] = None
|
|
185
|
+
partial_credit_components: Optional[dict] = None
|
|
186
|
+
reason: Optional[str] = None
|
|
187
|
+
|
|
188
|
+
def __bool__(self) -> bool:
|
|
189
|
+
return self.equivalent is True
|
|
190
|
+
|
|
191
|
+
def to_dict(self) -> dict:
|
|
192
|
+
"""Serialise to a JSON-compatible dict (all fields, field names as keys)."""
|
|
193
|
+
return {
|
|
194
|
+
"equivalent": self.equivalent,
|
|
195
|
+
"method_used": self.method_used,
|
|
196
|
+
"syntax_equal": self.syntax_equal,
|
|
197
|
+
"structurally_equal": self.structurally_equal,
|
|
198
|
+
"aligned_equal": self.aligned_equal,
|
|
199
|
+
"logically_equivalent": self.logically_equivalent,
|
|
200
|
+
"counterexample": self.counterexample,
|
|
201
|
+
"partial_credit": self.partial_credit,
|
|
202
|
+
"partial_credit_components": self.partial_credit_components,
|
|
203
|
+
"reason": self.reason,
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def _has_object_quantifier(node: Node) -> bool:
|
|
208
|
+
"""True iff ``node`` binds a first-order object variable anywhere."""
|
|
209
|
+
return any(isinstance(n, (Quantifier, SortedQuantifier)) for n in node.walk())
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _solver_tristate(f1: Node, f2: Node, timeout: int, frame: str, systems,
|
|
213
|
+
converse_axioms: tuple = ()):
|
|
214
|
+
"""Return ``(verdict, counterexample, reason)`` for genuine logical equivalence.
|
|
215
|
+
|
|
216
|
+
``reason`` is ``None`` unless the solver level REFUSED the input (a
|
|
217
|
+
``NotImplementedError`` from the translation, which says why: a family with no
|
|
218
|
+
first-order image, a numeral and a constant that are one symbol): then it is the
|
|
219
|
+
text of the refusal and the verdict is ``None``.
|
|
220
|
+
|
|
221
|
+
Classical route: Z3 on ``¬(φ ↔ ψ)`` — ``unsat`` proves equivalence, ``sat``
|
|
222
|
+
refutes it (the model is the counterexample), ``unknown`` stays ``None``.
|
|
223
|
+
Many-sorted input: ``sort_axioms(f1, f2)`` — every sort of either formula
|
|
224
|
+
is non-empty, and every sorted constant ``c:S`` of either formula is in
|
|
225
|
+
``S`` — is asserted as extra, UNNEGATED premises alongside ``¬(φ ↔ ψ)``:
|
|
226
|
+
the same soundness fix :mod:`unicode_logic_kit.atp.z3_models`/
|
|
227
|
+
``atp.protocol.Z3Backend`` apply, needed here for the identical reason.
|
|
228
|
+
MSFOL, by convention, never gives a sort an empty universe, and a sorted
|
|
229
|
+
constant denotes an element of its sort, but ``to_z3()``'s relativisation
|
|
230
|
+
alone carries neither guarantee (see the classical-reasoning guide's
|
|
231
|
+
many-sorted section). The facts of BOTH formulas are asserted: a legal
|
|
232
|
+
structure fixes ``c`` in ``S`` for either side. Empty for an unsorted pair,
|
|
233
|
+
so behaviour there is unchanged.
|
|
234
|
+
``converse_axioms`` (see :mod:`unicode_logic_kit.eval.converses`) are
|
|
235
|
+
asserted the same way — extra, UNNEGATED premises alongside
|
|
236
|
+
``¬(φ ↔ ψ)`` — before that negated goal is added, so ``solver.add`` sees
|
|
237
|
+
every premise (sort facts AND converse) ahead of the goal it bridges.
|
|
238
|
+
Empty by default, so behaviour is unchanged unless a caller opts in.
|
|
239
|
+
Modal route: ``modal_decide(Iff(φ, ψ))`` for the propositional fragment
|
|
240
|
+
(tri-state; a Kripke counter-model witnesses refutation); quantified or
|
|
241
|
+
tableau-rejected modal formulas fall back to ``qml_equivalent`` — sound but
|
|
242
|
+
bounded-incomplete, so its ``False`` is reported as ``None`` (not proven),
|
|
243
|
+
never as a refutation. ``converse_axioms`` has no modal route at all —
|
|
244
|
+
see the ``NotImplementedError`` below, raised before either modal branch
|
|
245
|
+
runs.
|
|
246
|
+
"""
|
|
247
|
+
from unicode_logic_kit.atp.modal_tableau import has_modal
|
|
248
|
+
|
|
249
|
+
iff = Iff(f1, f2)
|
|
250
|
+
|
|
251
|
+
if has_modal(iff):
|
|
252
|
+
if converse_axioms:
|
|
253
|
+
raise NotImplementedError(
|
|
254
|
+
"equivalent: converses is not supported for modal formulas "
|
|
255
|
+
"-- declared converse bridging axioms are only honoured by "
|
|
256
|
+
"the classical/MSFOL Z3 route")
|
|
257
|
+
from unicode_logic_kit.fol.qml import qml_equivalent
|
|
258
|
+
|
|
259
|
+
if not _has_object_quantifier(iff):
|
|
260
|
+
from unicode_logic_kit.atp.modal_tableau import (
|
|
261
|
+
modal_decide, modal_countermodel,
|
|
262
|
+
)
|
|
263
|
+
try:
|
|
264
|
+
status = modal_decide(iff, frame=frame, systems=systems)
|
|
265
|
+
except NotImplementedError:
|
|
266
|
+
status = None # fragment the tableau rejects
|
|
267
|
+
if status == "valid":
|
|
268
|
+
return True, None, None
|
|
269
|
+
if status == "invalid":
|
|
270
|
+
cm = modal_countermodel(iff, frame=frame, systems=systems)
|
|
271
|
+
witness = {"kind": "kripke", "repr": repr(cm)} if cm is not None else None
|
|
272
|
+
return False, witness, None
|
|
273
|
+
# "unknown" or rejected: fall through to the QML embedding.
|
|
274
|
+
try:
|
|
275
|
+
proved = qml_equivalent(f1, f2, frame=frame, systems=systems,
|
|
276
|
+
timeout=timeout)
|
|
277
|
+
except NotImplementedError as exc:
|
|
278
|
+
return None, None, str(exc)
|
|
279
|
+
return (True, None, None) if proved else (None, None, None)
|
|
280
|
+
|
|
281
|
+
# Classical route: tri-state Z3 (deliberately NOT formulas_are_equivalent,
|
|
282
|
+
# which collapses unknown to False — unusable as a metric verdict).
|
|
283
|
+
from z3 import Solver, Not as _ZNot, sat, unsat
|
|
284
|
+
|
|
285
|
+
from unicode_logic_kit.fol.nodes import Z3Env
|
|
286
|
+
|
|
287
|
+
try:
|
|
288
|
+
env = Z3Env() # one environment for both formulas and every extra assertion
|
|
289
|
+
phi, psi = f1.to_z3(env), f2.to_z3(env)
|
|
290
|
+
sort_facts = [axiom.to_z3(env) for axiom in sort_axioms(f1, f2)]
|
|
291
|
+
converses_z3 = [axiom.to_z3(env) for axiom in converse_axioms]
|
|
292
|
+
except NotImplementedError as exc:
|
|
293
|
+
return None, None, str(exc) # no Z3 image for this input: the refusal says why
|
|
294
|
+
solver = Solver()
|
|
295
|
+
solver.set("timeout", timeout)
|
|
296
|
+
solver.set("random_seed", 42)
|
|
297
|
+
for axiom in sort_facts:
|
|
298
|
+
solver.add(axiom)
|
|
299
|
+
for axiom in converses_z3:
|
|
300
|
+
solver.add(axiom)
|
|
301
|
+
solver.add(_ZNot(phi == psi))
|
|
302
|
+
res = solver.check()
|
|
303
|
+
if res == unsat:
|
|
304
|
+
return True, None, None
|
|
305
|
+
if res == sat:
|
|
306
|
+
from unicode_logic_kit.atp.z3_models import model_assignment
|
|
307
|
+
assignment = model_assignment(solver.model())
|
|
308
|
+
return False, {"kind": "z3_model", "assignment": assignment}, None
|
|
309
|
+
return None, None, None
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def _partial_credit(prediction: Node, reference: Node, verdict: Optional[bool],
|
|
313
|
+
max_norm_distance: float):
|
|
314
|
+
"""Return ``(partial_credit, partial_credit_components)`` for the rubric.
|
|
315
|
+
|
|
316
|
+
``verdict`` is the headline equivalence verdict for the call site — which,
|
|
317
|
+
for both places this is invoked (``method="solver"`` and the auto ladder's
|
|
318
|
+
solver fallthrough), IS the solver's tri-state verdict, i.e. the same
|
|
319
|
+
value that ends up in ``EquivalenceResult.equivalent`` /
|
|
320
|
+
``.logically_equivalent``. See the ``EquivalenceResult`` docstring for the
|
|
321
|
+
full rubric this implements; this function is deliberately dumb (no
|
|
322
|
+
control flow beyond the ``verdict is True`` short-circuit) so the rubric
|
|
323
|
+
stays easy to audit against that docstring.
|
|
324
|
+
"""
|
|
325
|
+
if verdict is True:
|
|
326
|
+
return 1.0, None
|
|
327
|
+
|
|
328
|
+
from .validate import is_wellformed
|
|
329
|
+
from .predicate_match import align_symbols, aligned_exact_match, _symbol_inventory
|
|
330
|
+
|
|
331
|
+
s1 = is_wellformed(prediction) and is_wellformed(reference)
|
|
332
|
+
|
|
333
|
+
aligned = align_symbols(prediction, reference, max_norm_distance)
|
|
334
|
+
ap, af, ac = _symbol_inventory(aligned)
|
|
335
|
+
rp, rf, rc = _symbol_inventory(reference)
|
|
336
|
+
s2 = ap == rp and af == rf and ac == rc
|
|
337
|
+
|
|
338
|
+
s3 = aligned_exact_match(prediction, reference, max_norm_distance)
|
|
339
|
+
|
|
340
|
+
s4 = verdict is True # never True here; see docstring
|
|
341
|
+
|
|
342
|
+
components = {"s1": s1, "s2": s2, "s3": s3, "s4": s4}
|
|
343
|
+
score = 0.25 * sum(components.values())
|
|
344
|
+
return score, components
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def equivalent(prediction: Node, reference: Node, *, method: str = "auto",
|
|
348
|
+
timeout: int = 10000, max_norm_distance: float = 0.6,
|
|
349
|
+
frame: str = "K", systems=None,
|
|
350
|
+
converses=None) -> EquivalenceResult:
|
|
351
|
+
"""Decide whether ``prediction`` and ``reference`` are equivalent.
|
|
352
|
+
|
|
353
|
+
Args:
|
|
354
|
+
prediction / reference: parsed formulas (any family the requested
|
|
355
|
+
level supports; the solver level covers classical FOL/MSFOL and
|
|
356
|
+
the modal family).
|
|
357
|
+
method: one of ``"exact"``, ``"canonical"``, ``"predicate_align"``,
|
|
358
|
+
``"solver"``, ``"auto"``. ``auto`` runs the ladder cheapest-first
|
|
359
|
+
and stops at the first ``True``; the solver only runs when the
|
|
360
|
+
structural levels all fail.
|
|
361
|
+
timeout: solver budget in milliseconds.
|
|
362
|
+
max_norm_distance: threshold for the ``predicate_align`` level (see
|
|
363
|
+
:func:`~unicode_logic_kit.eval.predicate_match.align_symbols`).
|
|
364
|
+
frame / systems: modal frame class and per-family systems, forwarded
|
|
365
|
+
to the modal deciders; ignored for classical formulas.
|
|
366
|
+
converses: an OPT-IN, default-``None`` sequence of
|
|
367
|
+
:data:`~unicode_logic_kit.eval.converses.ConverseDeclaration`
|
|
368
|
+
(``(a_key, b_key, permutation)`` triples — see that module) —
|
|
369
|
+
e.g. ``[(("LovedBy", 2), ("Loves", 2), (1, 0))]`` to bridge
|
|
370
|
+
``LovedBy(x, y) ↔ Loves(y, x)``. ``None``/empty is a strict no-op
|
|
371
|
+
(byte-identical behaviour to calling without this argument).
|
|
372
|
+
Non-empty requires ``method`` in ``{"solver", "auto"}`` —
|
|
373
|
+
``ValueError`` immediately otherwise, since the structural levels
|
|
374
|
+
never reach the solver and would silently ignore a declared
|
|
375
|
+
axiom instead of honouring it. When the solver level runs with
|
|
376
|
+
non-empty ``converses``, the axioms are asserted as extra
|
|
377
|
+
premises and the result's ``method_used`` is
|
|
378
|
+
``"solver_modulo_converses"`` instead of ``"solver"`` — its own
|
|
379
|
+
category, never merged into plain solver verdicts. A modal
|
|
380
|
+
formula pair with non-empty ``converses`` raises
|
|
381
|
+
:class:`NotImplementedError` (no modal bridging route exists).
|
|
382
|
+
|
|
383
|
+
Returns:
|
|
384
|
+
An :class:`EquivalenceResult`. Note the asymmetry of the levels: the
|
|
385
|
+
structural levels can only *establish* equivalence (their ``False``
|
|
386
|
+
just means "this quotient did not close the gap"), so only the solver
|
|
387
|
+
level can make the headline verdict ``False`` — and only with a
|
|
388
|
+
counterexample in hand.
|
|
389
|
+
"""
|
|
390
|
+
if method not in _METHODS:
|
|
391
|
+
raise ValueError(f"equivalent: unknown method {method!r} (use one of {_METHODS})")
|
|
392
|
+
|
|
393
|
+
from .canonical import exact_match
|
|
394
|
+
from .predicate_match import aligned_exact_match
|
|
395
|
+
|
|
396
|
+
axioms: tuple = ()
|
|
397
|
+
if converses:
|
|
398
|
+
if method in ("exact", "canonical", "predicate_align"):
|
|
399
|
+
raise ValueError(
|
|
400
|
+
f"equivalent: converses requires method in "
|
|
401
|
+
f"{{'solver', 'auto'}} (got {method!r}) — a declared "
|
|
402
|
+
"converse axiom is honoured only by the solver level; a "
|
|
403
|
+
"structural method would silently ignore it")
|
|
404
|
+
from .converses import converse_axioms as _converse_axioms
|
|
405
|
+
axioms = _converse_axioms(converses)
|
|
406
|
+
|
|
407
|
+
if method == "exact":
|
|
408
|
+
eq = prediction == reference
|
|
409
|
+
return EquivalenceResult(
|
|
410
|
+
equivalent=True if eq else None, method_used="exact", syntax_equal=eq)
|
|
411
|
+
|
|
412
|
+
if method == "canonical":
|
|
413
|
+
eq = exact_match(prediction, reference)
|
|
414
|
+
return EquivalenceResult(
|
|
415
|
+
equivalent=True if eq else None, method_used="canonical",
|
|
416
|
+
structurally_equal=eq)
|
|
417
|
+
|
|
418
|
+
if method == "predicate_align":
|
|
419
|
+
eq = aligned_exact_match(prediction, reference, max_norm_distance)
|
|
420
|
+
return EquivalenceResult(
|
|
421
|
+
equivalent=True if eq else None, method_used="predicate_align",
|
|
422
|
+
aligned_equal=eq)
|
|
423
|
+
|
|
424
|
+
if method == "solver":
|
|
425
|
+
verdict, cex, reason = _solver_tristate(prediction, reference, timeout, frame,
|
|
426
|
+
systems, axioms)
|
|
427
|
+
pc, components = _partial_credit(prediction, reference, verdict, max_norm_distance)
|
|
428
|
+
method_used = "solver_modulo_converses" if axioms else "solver"
|
|
429
|
+
return EquivalenceResult(
|
|
430
|
+
equivalent=verdict, method_used=method_used,
|
|
431
|
+
logically_equivalent=verdict, counterexample=cex,
|
|
432
|
+
partial_credit=pc, partial_credit_components=components, reason=reason)
|
|
433
|
+
|
|
434
|
+
# auto: cheapest first, stop at the first True; solver decides the rest.
|
|
435
|
+
# Every branch below computes partial_credit (method="auto" is one of the
|
|
436
|
+
# two rubric-eligible methods) — the early-True branches all score 1.0
|
|
437
|
+
# per the rubric's headline short-circuit.
|
|
438
|
+
syntax_equal = prediction == reference
|
|
439
|
+
if syntax_equal:
|
|
440
|
+
return EquivalenceResult(
|
|
441
|
+
equivalent=True, method_used="exact", syntax_equal=True,
|
|
442
|
+
partial_credit=1.0)
|
|
443
|
+
|
|
444
|
+
structurally_equal = exact_match(prediction, reference)
|
|
445
|
+
if structurally_equal:
|
|
446
|
+
return EquivalenceResult(
|
|
447
|
+
equivalent=True, method_used="canonical",
|
|
448
|
+
syntax_equal=False, structurally_equal=True,
|
|
449
|
+
partial_credit=1.0)
|
|
450
|
+
|
|
451
|
+
aligned_equal = aligned_exact_match(prediction, reference, max_norm_distance)
|
|
452
|
+
if aligned_equal:
|
|
453
|
+
return EquivalenceResult(
|
|
454
|
+
equivalent=True, method_used="predicate_align",
|
|
455
|
+
syntax_equal=False, structurally_equal=False, aligned_equal=True,
|
|
456
|
+
partial_credit=1.0)
|
|
457
|
+
|
|
458
|
+
verdict, cex, reason = _solver_tristate(prediction, reference, timeout, frame,
|
|
459
|
+
systems, axioms)
|
|
460
|
+
pc, components = _partial_credit(prediction, reference, verdict, max_norm_distance)
|
|
461
|
+
method_used = "solver_modulo_converses" if axioms else "solver"
|
|
462
|
+
return EquivalenceResult(
|
|
463
|
+
equivalent=verdict, method_used=method_used,
|
|
464
|
+
syntax_equal=False, structurally_equal=False, aligned_equal=False,
|
|
465
|
+
logically_equivalent=verdict, counterexample=cex,
|
|
466
|
+
partial_credit=pc, partial_credit_components=components, reason=reason)
|