unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,458 @@
|
|
|
1
|
+
"""A HuggingFace ``evaluate``-compatible NL→FOL metric.
|
|
2
|
+
|
|
3
|
+
The `evaluate <https://github.com/huggingface/evaluate>`_ library ships
|
|
4
|
+
metrics for classification, translation (BLEU/ROUGE/...), and generic
|
|
5
|
+
sequence tasks, but nothing that understands first-order logic: two
|
|
6
|
+
syntactically different formulas can denote the very same claim (``P ∧
|
|
7
|
+
Q`` / ``Q ∧ P``; ``Winner(x)`` / ``IsWinner(x)``), and a plain
|
|
8
|
+
string-equality metric scores that as wrong. This module plugs that gap by
|
|
9
|
+
wrapping :func:`unicode_logic_kit.eval.equivalence.equivalent` — the kit's
|
|
10
|
+
graded structural→solver equivalence ladder — as a metric.
|
|
11
|
+
|
|
12
|
+
Per-pair semantics
|
|
13
|
+
-------------------
|
|
14
|
+
Every ``(prediction, reference)`` row is scored **independently**: the two
|
|
15
|
+
strings are parsed on their own (via
|
|
16
|
+
:func:`unicode_logic_kit.api.parse_any`, which auto-detects the dialect —
|
|
17
|
+
the kit's unicode surface syntax, TPTP, Prover9, LaTeX, or SMT-LIB) and then
|
|
18
|
+
compared with :func:`~unicode_logic_kit.eval.equivalence.equivalent`. There is
|
|
19
|
+
no cross-row state (no shared vocabulary, no corpus-level normalisation) —
|
|
20
|
+
this mirrors how the kit's other evaluation entry points
|
|
21
|
+
(:func:`~unicode_logic_kit.eval.batch.batch_decide`) treat a batch as
|
|
22
|
+
independent tasks, and it means a caller can freely shuffle, subset, or
|
|
23
|
+
parallelise rows without changing any pair's score.
|
|
24
|
+
|
|
25
|
+
What "equivalent" means
|
|
26
|
+
-------------------------
|
|
27
|
+
See :mod:`unicode_logic_kit.eval.equivalence` for the full ladder — in
|
|
28
|
+
short, ``exact`` (AST ``==``) → ``canonical`` (α-renaming,
|
|
29
|
+
commutativity/associativity, double-negation) → ``predicate_align``
|
|
30
|
+
(vocabulary renaming on top of canonical) → ``solver`` (genuine logical
|
|
31
|
+
equivalence via Z3 or the modal decider, TRI-STATE: proved / refuted /
|
|
32
|
+
unknown). ``method="auto"`` (this module's default, matching
|
|
33
|
+
:func:`equivalent`'s own default) runs the ladder cheapest-first and stops
|
|
34
|
+
at the first level that proves equivalence, falling through to the solver
|
|
35
|
+
only when every structural level fails.
|
|
36
|
+
|
|
37
|
+
The honesty contract
|
|
38
|
+
----------------------
|
|
39
|
+
The solver's tri-state verdict is the reason this module reports several
|
|
40
|
+
numbers instead of one "accuracy":
|
|
41
|
+
|
|
42
|
+
* ``equivalence_accuracy`` counts ONLY a definitive ``equivalent is True``
|
|
43
|
+
as correct. A refutation (``False``, with a counterexample) and an
|
|
44
|
+
undecided solver call (``None``, the budget/completeness limit was hit
|
|
45
|
+
without an answer either way) are both counted as *not correct* — but
|
|
46
|
+
they are NOT THE SAME THING, and collapsing them together (the defect
|
|
47
|
+
:func:`unicode_logic_kit.atp.formulas_are_equivalent` has, which
|
|
48
|
+
:func:`equivalent` exists to avoid) would silently misreport how often
|
|
49
|
+
the metric actually *knows* the prediction is wrong.
|
|
50
|
+
* ``solver_unknown_rate`` makes that hidden mass visible: the fraction of
|
|
51
|
+
pairs where the ladder reached the solver level (i.e. no structural level
|
|
52
|
+
decided the pair, so :class:`~unicode_logic_kit.eval.equivalence
|
|
53
|
+
.EquivalenceResult`'s ``method_used`` is ``"solver"``) and the solver
|
|
54
|
+
came back ``None``. It is deliberately reported alongside
|
|
55
|
+
``equivalence_accuracy`` rather than folded into it, so a consumer can
|
|
56
|
+
compute honest bounds: the true accuracy lies in
|
|
57
|
+
``[equivalence_accuracy, equivalence_accuracy + solver_unknown_rate]``.
|
|
58
|
+
(With ``method`` values other than ``"auto"``/``"solver"`` the solver
|
|
59
|
+
never runs at all, so this rate is always ``0.0`` — that is not a claim
|
|
60
|
+
every pair was decided, just that this particular signal has nothing to
|
|
61
|
+
report; see :func:`compute_fol_metrics`'s docstring.)
|
|
62
|
+
* A parse failure on either side of a pair is scored ``0.0`` for that pair
|
|
63
|
+
(it cannot be "equivalent" to anything, having never become a formula)
|
|
64
|
+
and is tallied into ``parse_failure_rate`` — never silently dropped
|
|
65
|
+
from the batch (which would inflate every other rate by shrinking the
|
|
66
|
+
denominator) and never counted as a solver "unknown" (it never reached
|
|
67
|
+
the solver, or any level of the ladder, at all).
|
|
68
|
+
"""
|
|
69
|
+
|
|
70
|
+
import importlib.util
|
|
71
|
+
from typing import List
|
|
72
|
+
|
|
73
|
+
from .equivalence import equivalent
|
|
74
|
+
|
|
75
|
+
__all__ = ["compute_fol_metrics", "FolEquivalence", "load"]
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
# ---------------------------------------------------------------------------
|
|
79
|
+
# compute_fol_metrics -- works with or without the `evaluate` package.
|
|
80
|
+
# ---------------------------------------------------------------------------
|
|
81
|
+
|
|
82
|
+
def _score_pair(prediction: str, reference: str, method: str, timeout_ms: int,
|
|
83
|
+
converses=None) -> dict:
|
|
84
|
+
"""Score ONE ``(prediction, reference)`` pair. Never raises.
|
|
85
|
+
|
|
86
|
+
Returns ``{"parse_failure", "syntax_equal", "equivalent_true",
|
|
87
|
+
"partial_credit", "solver_unknown", "method_used"}`` (all bool except
|
|
88
|
+
``partial_credit``, a float, and ``method_used``, a str or ``None``). A
|
|
89
|
+
parse failure on either side reports every bool/float field at its floor
|
|
90
|
+
(``False`` / ``0.0``, ``method_used=None``) — see the module docstring's
|
|
91
|
+
honesty-contract section for why that floor, not a skip, is the correct
|
|
92
|
+
score for text that never became a formula.
|
|
93
|
+
|
|
94
|
+
``converses`` (see :mod:`unicode_logic_kit.eval.converses`) is forwarded
|
|
95
|
+
to :func:`~unicode_logic_kit.eval.equivalence.equivalent` unchanged;
|
|
96
|
+
``method_used`` is carried back out so :func:`compute_fol_metrics` can
|
|
97
|
+
compute ``converse_matched_rate`` without recomputing the verdict.
|
|
98
|
+
|
|
99
|
+
``syntax_equal`` is computed directly as ``prediction_formula ==
|
|
100
|
+
reference_formula`` rather than read off
|
|
101
|
+
``EquivalenceResult.syntax_equal``: that field is only populated by the
|
|
102
|
+
``"exact"`` and ``"auto"`` methods (see ``equivalence.py``'s
|
|
103
|
+
``equivalent()`` — the ``"canonical"``/``"predicate_align"``/``"solver"``
|
|
104
|
+
branches never set it), so reading it directly would silently under-count
|
|
105
|
+
``exact_match`` for every other ``method`` value. Computing it ourselves
|
|
106
|
+
keeps ``exact_match`` well-defined for any ``method``.
|
|
107
|
+
|
|
108
|
+
``partial_credit`` reads ``EquivalenceResult.partial_credit`` when the
|
|
109
|
+
ladder computed one (``method="auto"``/``"solver"`` only — see that
|
|
110
|
+
field's docstring) and floors to ``0.0`` otherwise: the metric cannot
|
|
111
|
+
report a score the ladder never computed, so "not computed" and "computed
|
|
112
|
+
as the minimum" must not be conflated into a fabricated high average, and
|
|
113
|
+
``0.0`` is the conservative choice consistent with the parse-failure floor
|
|
114
|
+
above.
|
|
115
|
+
"""
|
|
116
|
+
from .. import api # lazy: avoid import-time cost/cycles (mirrors eval/batch.py)
|
|
117
|
+
|
|
118
|
+
pred_parsed = api.parse_any(prediction)
|
|
119
|
+
ref_parsed = api.parse_any(reference)
|
|
120
|
+
if not pred_parsed.ok or not ref_parsed.ok:
|
|
121
|
+
return {
|
|
122
|
+
"parse_failure": True,
|
|
123
|
+
"syntax_equal": False,
|
|
124
|
+
"equivalent_true": False,
|
|
125
|
+
"partial_credit": 0.0,
|
|
126
|
+
"solver_unknown": False,
|
|
127
|
+
"method_used": None,
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
result = equivalent(pred_parsed.formula, ref_parsed.formula,
|
|
131
|
+
method=method, timeout=timeout_ms, converses=converses)
|
|
132
|
+
|
|
133
|
+
# method_used in {"solver", "solver_modulo_converses"} happens exactly
|
|
134
|
+
# when the ladder reached the solver level -- both for method="solver"
|
|
135
|
+
# directly and for the "auto" ladder's fallthrough (see equivalent()'s
|
|
136
|
+
# auto branch: it ALWAYS labels the solver fallthrough this way, whether
|
|
137
|
+
# the tri-state verdict came back True, False, or None) -- the latter
|
|
138
|
+
# tag only ever appears when `converses` was non-empty.
|
|
139
|
+
solver_unknown = (result.method_used in ("solver", "solver_modulo_converses")
|
|
140
|
+
and result.equivalent is None)
|
|
141
|
+
|
|
142
|
+
return {
|
|
143
|
+
"parse_failure": False,
|
|
144
|
+
"syntax_equal": pred_parsed.formula == ref_parsed.formula,
|
|
145
|
+
"equivalent_true": result.equivalent is True,
|
|
146
|
+
"partial_credit": result.partial_credit if result.partial_credit is not None else 0.0,
|
|
147
|
+
"solver_unknown": solver_unknown,
|
|
148
|
+
"method_used": result.method_used,
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def compute_fol_metrics(predictions: List[str], references: List[str], *,
|
|
153
|
+
method: str = "auto", timeout_ms: int = 10000,
|
|
154
|
+
converses=None) -> dict:
|
|
155
|
+
"""Score a batch of NL→FOL predictions against references, per-pair.
|
|
156
|
+
|
|
157
|
+
Args:
|
|
158
|
+
predictions: predicted formula strings, any dialect
|
|
159
|
+
:func:`unicode_logic_kit.api.parse_any` can detect.
|
|
160
|
+
references: reference (gold) formula strings, same dialect rules.
|
|
161
|
+
Must be the same length as ``predictions``.
|
|
162
|
+
method: forwarded to :func:`~unicode_logic_kit.eval.equivalence
|
|
163
|
+
.equivalent` for every pair — one of ``"exact"``,
|
|
164
|
+
``"canonical"``, ``"predicate_align"``, ``"solver"``, ``"auto"``
|
|
165
|
+
(the default: the full ladder, cheapest level first).
|
|
166
|
+
timeout_ms: forwarded as ``equivalent()``'s ``timeout`` (milliseconds)
|
|
167
|
+
for every pair's solver call.
|
|
168
|
+
converses: OPT-IN, default ``None`` — forwarded unchanged to
|
|
169
|
+
:func:`~unicode_logic_kit.eval.equivalence.equivalent` for every
|
|
170
|
+
pair (see that function's ``converses`` parameter and
|
|
171
|
+
:mod:`unicode_logic_kit.eval.converses`). ``None``/empty leaves the
|
|
172
|
+
returned dict's KEY SET unchanged (required — see
|
|
173
|
+
``tests/test_metric_hf.py``'s full-dict-equality asserts); a
|
|
174
|
+
non-empty sequence adds the ``converse_matched_rate`` key below.
|
|
175
|
+
|
|
176
|
+
Returns:
|
|
177
|
+
A dict with six keys (seven when ``converses`` is non-empty):
|
|
178
|
+
|
|
179
|
+
* ``exact_match`` — fraction of pairs whose parsed formulas are
|
|
180
|
+
AST-equal (``==``); ``0.0`` for an unparseable pair.
|
|
181
|
+
* ``equivalence_accuracy`` — fraction of pairs where
|
|
182
|
+
``equivalent(...).equivalent is True``. See the module docstring's
|
|
183
|
+
honesty-contract section: this is NOT ``1 - (fraction refuted)``,
|
|
184
|
+
because undecided pairs are excluded from the numerator without
|
|
185
|
+
being counted as refuted either.
|
|
186
|
+
* ``mean_partial_credit`` — mean of
|
|
187
|
+
``equivalent(...).partial_credit`` over all pairs, treating an
|
|
188
|
+
unparseable pair or a ``partial_credit`` the requested ``method``
|
|
189
|
+
never computes (see that field's docstring) as ``0.0``.
|
|
190
|
+
* ``parse_failure_rate`` — fraction of pairs where either side
|
|
191
|
+
failed to parse.
|
|
192
|
+
* ``solver_unknown_rate`` — fraction of pairs where the ladder
|
|
193
|
+
reached the solver level and it returned ``None`` (undecided).
|
|
194
|
+
Always ``0.0`` for ``method`` values that never invoke the solver
|
|
195
|
+
(``"exact"``/``"canonical"``/``"predicate_align"``).
|
|
196
|
+
* ``n`` — the batch size (``len(predictions)``).
|
|
197
|
+
* ``converse_matched_rate`` — ONLY present when ``converses`` is
|
|
198
|
+
non-empty: the fraction of pairs where the solver level ran WITH
|
|
199
|
+
the declared axioms (``method_used == "solver_modulo_converses"``)
|
|
200
|
+
AND proved equivalence (``equivalent is True``). A separately
|
|
201
|
+
visible, subtractable slice of ``equivalence_accuracy`` — never
|
|
202
|
+
folded into it silently, matching this module's honesty-contract
|
|
203
|
+
convention for ``solver_unknown_rate`` above.
|
|
204
|
+
|
|
205
|
+
Raises:
|
|
206
|
+
ValueError: ``predictions`` and ``references`` have different
|
|
207
|
+
lengths; a malformed ``converses`` declaration (checked
|
|
208
|
+
unconditionally, including for an empty batch); or a non-empty
|
|
209
|
+
``converses`` combined with a ``method`` that cannot honour it
|
|
210
|
+
(``"exact"``/``"canonical"``/``"predicate_align"`` — also
|
|
211
|
+
checked unconditionally, matching :func:`~unicode_logic_kit.eval
|
|
212
|
+
.equivalence.equivalent`'s own check for ``n >= 1``).
|
|
213
|
+
"""
|
|
214
|
+
if len(predictions) != len(references):
|
|
215
|
+
raise ValueError(
|
|
216
|
+
"compute_fol_metrics: predictions and references must have the "
|
|
217
|
+
f"same length (got {len(predictions)} and {len(references)})")
|
|
218
|
+
|
|
219
|
+
if converses:
|
|
220
|
+
# Validate unconditionally, even for an empty batch: _score_pair (the
|
|
221
|
+
# only other call site that would reach validate_converses, via
|
|
222
|
+
# equivalent()) never runs when n == 0, so without this a malformed
|
|
223
|
+
# declaration would silently pass through the n == 0 branch below
|
|
224
|
+
# instead of raising -- see this function's own documented contract
|
|
225
|
+
# ("Raises: ValueError ... a malformed converses declaration").
|
|
226
|
+
from .converses import validate_converses
|
|
227
|
+
validate_converses(converses)
|
|
228
|
+
# Mirror equivalent()'s own method-gating check (equivalence.py:
|
|
229
|
+
# "converses requires method in {'solver', 'auto'}") unconditionally
|
|
230
|
+
# too, for the same reason: that check normally runs inside
|
|
231
|
+
# equivalent() via _score_pair, which never executes when n == 0, so
|
|
232
|
+
# without this a solver-incompatible method combined with a
|
|
233
|
+
# (structurally valid) converses declaration would raise for n >= 1
|
|
234
|
+
# but silently return zeroed metrics for n == 0 -- same declaration,
|
|
235
|
+
# same method, different n, different behaviour. Kept as an exact
|
|
236
|
+
# duplicate of equivalent()'s wording (not just "the same class of
|
|
237
|
+
# error") so a caller sees one consistent message regardless of
|
|
238
|
+
# which code path raised it.
|
|
239
|
+
if method in ("exact", "canonical", "predicate_align"):
|
|
240
|
+
raise ValueError(
|
|
241
|
+
f"equivalent: converses requires method in "
|
|
242
|
+
f"{{'solver', 'auto'}} (got {method!r}) — a declared "
|
|
243
|
+
"converse axiom is honoured only by the solver level; a "
|
|
244
|
+
"structural method would silently ignore it")
|
|
245
|
+
|
|
246
|
+
n = len(predictions)
|
|
247
|
+
if n == 0:
|
|
248
|
+
result = {
|
|
249
|
+
"exact_match": 0.0,
|
|
250
|
+
"equivalence_accuracy": 0.0,
|
|
251
|
+
"mean_partial_credit": 0.0,
|
|
252
|
+
"parse_failure_rate": 0.0,
|
|
253
|
+
"solver_unknown_rate": 0.0,
|
|
254
|
+
"n": 0,
|
|
255
|
+
}
|
|
256
|
+
if converses:
|
|
257
|
+
result["converse_matched_rate"] = 0.0
|
|
258
|
+
return result
|
|
259
|
+
|
|
260
|
+
scores = [_score_pair(p, r, method, timeout_ms, converses)
|
|
261
|
+
for p, r in zip(predictions, references)]
|
|
262
|
+
|
|
263
|
+
result = {
|
|
264
|
+
"exact_match": sum(s["syntax_equal"] for s in scores) / n,
|
|
265
|
+
"equivalence_accuracy": sum(s["equivalent_true"] for s in scores) / n,
|
|
266
|
+
"mean_partial_credit": sum(s["partial_credit"] for s in scores) / n,
|
|
267
|
+
"parse_failure_rate": sum(s["parse_failure"] for s in scores) / n,
|
|
268
|
+
"solver_unknown_rate": sum(s["solver_unknown"] for s in scores) / n,
|
|
269
|
+
"n": n,
|
|
270
|
+
}
|
|
271
|
+
if converses:
|
|
272
|
+
result["converse_matched_rate"] = sum(
|
|
273
|
+
s["method_used"] == "solver_modulo_converses" and s["equivalent_true"]
|
|
274
|
+
for s in scores) / n
|
|
275
|
+
return result
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
# ---------------------------------------------------------------------------
|
|
279
|
+
# FolEquivalence -- the evaluate.Metric wrapper (optional dependency).
|
|
280
|
+
# ---------------------------------------------------------------------------
|
|
281
|
+
#
|
|
282
|
+
# `evaluate` is NOT a hard dependency of this module: importing metric_hf.py
|
|
283
|
+
# must succeed on a machine that never installed it (mirroring how
|
|
284
|
+
# atp/cvc5_backend.py's Cvc5Backend stays importable without cvc5 -- see that
|
|
285
|
+
# module's docstring). Only *instantiating* FolEquivalence requires it -- and,
|
|
286
|
+
# unlike the eager `try: import evaluate / except ImportError` this used to
|
|
287
|
+
# do, that import must not happen just because metric_hf.py itself was
|
|
288
|
+
# imported (`evaluate` drags in `datasets`, together well over a second of
|
|
289
|
+
# import time -- see the module docstring's own claim, which this section
|
|
290
|
+
# makes actually true). `_HAS_EVALUATE` below is pure discovery -- no import
|
|
291
|
+
# -- mirroring atp/cvc5_backend.py's `Cvc5Backend.available()`.
|
|
292
|
+
|
|
293
|
+
_HAS_EVALUATE = (importlib.util.find_spec("evaluate") is not None and
|
|
294
|
+
importlib.util.find_spec("datasets") is not None)
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
_INSTALL_HINT = (
|
|
298
|
+
"FolEquivalence requires the optional 'evaluate' package "
|
|
299
|
+
"(pip install evaluate) -- it is not installed in this environment. "
|
|
300
|
+
"unicode_logic_kit.eval.metric_hf.compute_fol_metrics(predictions, "
|
|
301
|
+
"references) gives the same scores without that dependency."
|
|
302
|
+
)
|
|
303
|
+
|
|
304
|
+
_DESCRIPTION = (
|
|
305
|
+
"Graded NL→FOL equivalence: scores each (prediction, reference) pair "
|
|
306
|
+
"with unicode_logic_kit's structural→solver equivalence ladder "
|
|
307
|
+
"(unicode_logic_kit.eval.equivalence.equivalent). Reports exact_match, "
|
|
308
|
+
"equivalence_accuracy (solver-tri-state-aware -- undecided pairs are "
|
|
309
|
+
"counted as neither correct nor refuted), mean_partial_credit, "
|
|
310
|
+
"parse_failure_rate, and solver_unknown_rate (the undecided mass, kept "
|
|
311
|
+
"separate from equivalence_accuracy on purpose -- see "
|
|
312
|
+
"unicode_logic_kit.eval.metric_hf's module docstring)."
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
_KWARGS_DESCRIPTION = """
|
|
316
|
+
Args:
|
|
317
|
+
predictions (list of str): predicted FOL formula strings.
|
|
318
|
+
references (list of str): reference (gold) FOL formula strings, same
|
|
319
|
+
length as `predictions`.
|
|
320
|
+
method (str, optional): equivalence-ladder level, one of "exact",
|
|
321
|
+
"canonical", "predicate_align", "solver", "auto" (default).
|
|
322
|
+
timeout_ms (int, optional): per-pair solver budget in milliseconds
|
|
323
|
+
(default 10000).
|
|
324
|
+
|
|
325
|
+
Returns:
|
|
326
|
+
exact_match (float): fraction of pairs with AST-equal parsed formulas.
|
|
327
|
+
equivalence_accuracy (float): fraction of pairs the ladder proved
|
|
328
|
+
equivalent (a definitive True only -- see the module docstring).
|
|
329
|
+
mean_partial_credit (float): mean heuristic partial-credit score.
|
|
330
|
+
parse_failure_rate (float): fraction of pairs where either side failed
|
|
331
|
+
to parse.
|
|
332
|
+
solver_unknown_rate (float): fraction of pairs where the solver was
|
|
333
|
+
reached and returned an undecided verdict.
|
|
334
|
+
n (int): batch size.
|
|
335
|
+
|
|
336
|
+
Examples:
|
|
337
|
+
>>> import unicode_logic_kit.eval.metric_hf as metric_hf
|
|
338
|
+
>>> m = metric_hf.load()
|
|
339
|
+
>>> m.compute(predictions=["P(a) ∧ Q(a)"], references=["Q(a) ∧ P(a)"])
|
|
340
|
+
{'exact_match': 0.0, 'equivalence_accuracy': 1.0, ...}
|
|
341
|
+
"""
|
|
342
|
+
|
|
343
|
+
# `FolEquivalence` itself is built lazily: the class body subclasses
|
|
344
|
+
# `evaluate.Metric`, so merely *defining* it needs `evaluate` (and, for
|
|
345
|
+
# `_info`'s `datasets.Features`, `datasets`) already imported. Building it
|
|
346
|
+
# eagerly at module scope -- even behind an `if _HAS_EVALUATE:` -- would
|
|
347
|
+
# import both packages the moment metric_hf.py is imported, which is
|
|
348
|
+
# exactly the cost this module's docstring already claims does NOT happen.
|
|
349
|
+
# `_build_fol_equivalence_class` does the real import (and only it pays that
|
|
350
|
+
# cost, once, on first use) and caches the resulting class in
|
|
351
|
+
# `_fol_equivalence_class`; the module-level `__getattr__` below (PEP 562)
|
|
352
|
+
# routes both `metric_hf.FolEquivalence` and
|
|
353
|
+
# `from ... import FolEquivalence` through it transparently.
|
|
354
|
+
|
|
355
|
+
_fol_equivalence_class = None
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def _build_fol_equivalence_class():
|
|
359
|
+
"""Import ``evaluate``/``datasets`` and return the real ``FolEquivalence``
|
|
360
|
+
class, building it once and caching it for every later call.
|
|
361
|
+
|
|
362
|
+
Raises:
|
|
363
|
+
ImportError: with the install hint, if ``evaluate``/``datasets`` are
|
|
364
|
+
not installed (or fail to import for any other reason) --
|
|
365
|
+
mirroring the message the pre-lazy fallback stub used to raise
|
|
366
|
+
from ``FolEquivalence.__init__``.
|
|
367
|
+
"""
|
|
368
|
+
global _fol_equivalence_class
|
|
369
|
+
if _fol_equivalence_class is not None:
|
|
370
|
+
return _fol_equivalence_class
|
|
371
|
+
|
|
372
|
+
try:
|
|
373
|
+
import evaluate as _evaluate
|
|
374
|
+
import datasets as _hf_datasets
|
|
375
|
+
except ImportError:
|
|
376
|
+
raise ImportError(_INSTALL_HINT) from None
|
|
377
|
+
|
|
378
|
+
class FolEquivalence(_evaluate.Metric):
|
|
379
|
+
"""``evaluate.Metric`` wrapper around :func:`compute_fol_metrics`.
|
|
380
|
+
|
|
381
|
+
A thin adapter: ``_compute`` delegates entirely to
|
|
382
|
+
:func:`compute_fol_metrics`, so this class adds nothing but the
|
|
383
|
+
``evaluate``/``datasets`` plumbing (feature schema, description,
|
|
384
|
+
the ``evaluate.load``-shaped ``load()`` entry point below) around
|
|
385
|
+
logic that already works standalone. See the module docstring for
|
|
386
|
+
what the returned numbers mean.
|
|
387
|
+
"""
|
|
388
|
+
|
|
389
|
+
def _info(self):
|
|
390
|
+
# Returns evaluate.MetricInfo. Deliberately NOT annotated with a
|
|
391
|
+
# `-> "_evaluate.MetricInfo"` string forward reference: `_evaluate`
|
|
392
|
+
# is a local variable of the enclosing `_build_fol_equivalence_class`
|
|
393
|
+
# (that is what keeps the import lazy -- see the section comment
|
|
394
|
+
# above), not a module global, so `typing.get_type_hints()` would
|
|
395
|
+
# raise NameError trying to resolve it. Nothing in this codebase
|
|
396
|
+
# or its test suite calls `get_type_hints` on this method; the
|
|
397
|
+
# type is documented here in prose instead.
|
|
398
|
+
return _evaluate.MetricInfo(
|
|
399
|
+
description=_DESCRIPTION,
|
|
400
|
+
citation="",
|
|
401
|
+
inputs_description=_KWARGS_DESCRIPTION,
|
|
402
|
+
features=_hf_datasets.Features({
|
|
403
|
+
"predictions": _hf_datasets.Value("string", id="sequence"),
|
|
404
|
+
"references": _hf_datasets.Value("string", id="sequence"),
|
|
405
|
+
}),
|
|
406
|
+
reference_urls=[
|
|
407
|
+
"https://unicode-logic-kit.readthedocs.io/",
|
|
408
|
+
],
|
|
409
|
+
)
|
|
410
|
+
|
|
411
|
+
def _compute(self, predictions, references, method: str = "auto",
|
|
412
|
+
timeout_ms: int = 10000) -> dict:
|
|
413
|
+
return compute_fol_metrics(list(predictions), list(references),
|
|
414
|
+
method=method, timeout_ms=timeout_ms)
|
|
415
|
+
|
|
416
|
+
# Built inside a function, so Python's default __qualname__ would be
|
|
417
|
+
# "_build_fol_equivalence_class.<locals>.FolEquivalence" -- reset it to
|
|
418
|
+
# the plain module-level name so anything that reads it (repr, pickling
|
|
419
|
+
# by reference, which resolves "module.qualname" via getattr and so
|
|
420
|
+
# works fine through __getattr__ below) sees the same identifier a
|
|
421
|
+
# module-scoped class definition would have had.
|
|
422
|
+
FolEquivalence.__qualname__ = "FolEquivalence"
|
|
423
|
+
|
|
424
|
+
_fol_equivalence_class = FolEquivalence
|
|
425
|
+
return _fol_equivalence_class
|
|
426
|
+
|
|
427
|
+
|
|
428
|
+
def __getattr__(name: str):
|
|
429
|
+
"""PEP 562 module-level attribute hook.
|
|
430
|
+
|
|
431
|
+
Only ``FolEquivalence`` is handled specially: accessing
|
|
432
|
+
``metric_hf.FolEquivalence`` (attribute access, ``from ... import
|
|
433
|
+
FolEquivalence``, or ``dir()``-following tools) builds -- and, on every
|
|
434
|
+
call after the first, just returns the cached -- real class via
|
|
435
|
+
:func:`_build_fol_equivalence_class`, so `evaluate`/`datasets` are only
|
|
436
|
+
ever imported when this name is actually touched, never merely because
|
|
437
|
+
this module was.
|
|
438
|
+
"""
|
|
439
|
+
if name == "FolEquivalence":
|
|
440
|
+
return _build_fol_equivalence_class()
|
|
441
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
def load():
|
|
445
|
+
"""Return a ready-to-use :class:`FolEquivalence` instance.
|
|
446
|
+
|
|
447
|
+
Mirrors the shape of ``evaluate.load("metric_name")`` for callers used
|
|
448
|
+
to that entry point. Raises :class:`ImportError` (with the install
|
|
449
|
+
hint) if the optional ``evaluate`` package is not installed -- use
|
|
450
|
+
:func:`compute_fol_metrics` directly in that case.
|
|
451
|
+
|
|
452
|
+
Not annotated ``-> "FolEquivalence"``: that name is only ever reachable
|
|
453
|
+
through the module-level ``__getattr__`` above (never bound as a true
|
|
454
|
+
module global -- that is what keeps it lazy), so a string forward
|
|
455
|
+
reference to it would raise NameError from ``typing.get_type_hints()``.
|
|
456
|
+
The return type is documented here in prose instead.
|
|
457
|
+
"""
|
|
458
|
+
return _build_fol_equivalence_class()()
|