unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
"""Adapter for the FOLIO dataset (Han et al., "FOLIO: Natural Language
|
|
2
|
+
Reasoning with First-Order Logic") — local JSONL only, no network access.
|
|
3
|
+
|
|
4
|
+
Source and verified schema
|
|
5
|
+
---------------------------
|
|
6
|
+
Canonical, unauthenticated source: https://github.com/Yale-LILY/FOLIO
|
|
7
|
+
(``data/v0.0/folio-train.jsonl`` / ``data/v0.0/folio-validation.jsonl``). A
|
|
8
|
+
mirror also exists at https://huggingface.co/datasets/yale-nlp/FOLIO, but
|
|
9
|
+
that mirror is access-gated (requires a Hugging Face login to download), so
|
|
10
|
+
it was NOT used to verify the schema below — everything here was verified
|
|
11
|
+
directly against the raw GitHub JSONL content.
|
|
12
|
+
|
|
13
|
+
Each verified ``v0.0`` JSONL line is a flat JSON object with exactly these
|
|
14
|
+
keys (confirmed by fetching and inspecting real rows of
|
|
15
|
+
``folio-validation.jsonl``):
|
|
16
|
+
|
|
17
|
+
* ``"premises"`` — ``list[str]``, the natural-language premises.
|
|
18
|
+
* ``"premises-FOL"`` — ``list[str]``, the SAME LENGTH and ORDER as
|
|
19
|
+
``"premises"``; each entry is that premise's gold FOL annotation, written
|
|
20
|
+
in this kit's own unicode surface syntax (``∀``/``∃``/``∧``/``∨``/``¬``/
|
|
21
|
+
``→``/``⊕`` etc. all appear in the verified sample and all parse under this
|
|
22
|
+
kit's default ``fol`` mode).
|
|
23
|
+
* ``"conclusion"`` — ``str``, the natural-language conclusion.
|
|
24
|
+
* ``"conclusion-FOL"`` — ``str``, its gold FOL annotation.
|
|
25
|
+
* ``"label"`` — ``str``, the gold entailment label. The verified
|
|
26
|
+
sample showed ``"True"`` and ``"Uncertain"``; the paper and repository
|
|
27
|
+
README additionally document ``"False"`` as the third value (NOT directly
|
|
28
|
+
observed in the two rows fetched for this verification — flagged here
|
|
29
|
+
rather than silently assumed).
|
|
30
|
+
|
|
31
|
+
The verified ``v0.0`` rows carry NO id field at all (no ``example-id``,
|
|
32
|
+
``story-id``, or ``source`` key was present in the fetched sample, contrary
|
|
33
|
+
to a fields description on the Hugging Face mirror's README, which this
|
|
34
|
+
adapter could not reach past the login gate to cross-check). This loader is
|
|
35
|
+
defensive rather than assuming either shape is final: it uses an
|
|
36
|
+
``"example-id"``/``"example_id"`` key WHEN PRESENT, and otherwise falls back
|
|
37
|
+
to a positional id ``f"folio:{line_no}"`` (0-based line number within the
|
|
38
|
+
file) so every example is still addressable. Deliberately NOT used for id
|
|
39
|
+
resolution: ``"story-id"``/``"story_id"`` — the verified sample shows several
|
|
40
|
+
consecutive rows sharing identical ``premises``/``premises-FOL`` (one story,
|
|
41
|
+
several conclusions), so a story id identifies a GROUP of examples, not one
|
|
42
|
+
example, and using it alone as an id would collide across that group. It is
|
|
43
|
+
still preserved (like every other unrecognised key) in ``meta`` when present.
|
|
44
|
+
|
|
45
|
+
License: **CC-BY-SA-4.0**, per the ``LICENSE`` file at the repository root
|
|
46
|
+
(https://github.com/Yale-LILY/FOLIO/blob/main/LICENSE, verified directly).
|
|
47
|
+
Share-alike: adaptations/redistributions of the data itself must carry a
|
|
48
|
+
compatible license and attribution — this loader only reads a LOCAL file the
|
|
49
|
+
caller already obtained, it does not redistribute or embed any FOLIO data.
|
|
50
|
+
|
|
51
|
+
This module never downloads anything — obtain the JSONL file yourself from
|
|
52
|
+
one of the sources above and pass its local path to :func:`load_folio`.
|
|
53
|
+
"""
|
|
54
|
+
|
|
55
|
+
import json
|
|
56
|
+
from pathlib import Path
|
|
57
|
+
from typing import FrozenSet, Iterator, Union
|
|
58
|
+
|
|
59
|
+
from ._base import DatasetExample, _register_dataset_info
|
|
60
|
+
|
|
61
|
+
__all__ = ["load_folio"]
|
|
62
|
+
|
|
63
|
+
_register_dataset_info(
|
|
64
|
+
"folio",
|
|
65
|
+
license="CC-BY-SA-4.0",
|
|
66
|
+
source_url="https://github.com/Yale-LILY/FOLIO",
|
|
67
|
+
citation_hint=(
|
|
68
|
+
"Han, Simin, et al. \"FOLIO: Natural Language Reasoning with "
|
|
69
|
+
"First-Order Logic.\" arXiv:2209.00840."
|
|
70
|
+
),
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _resolve_id(record: dict, line_no: int) -> str:
|
|
75
|
+
"""The record's own per-example id field if present, else a positional
|
|
76
|
+
fallback.
|
|
77
|
+
|
|
78
|
+
Tries both the hyphenated spelling documented in the field-description
|
|
79
|
+
text and the underscore spelling a JSON-key-safe re-export might use;
|
|
80
|
+
verified real data has neither (see module docstring), so the fallback
|
|
81
|
+
path is what actually fires against the canonical GitHub JSONL today.
|
|
82
|
+
Deliberately does NOT consult ``"story-id"``/``"story_id"`` — see the
|
|
83
|
+
module docstring for why that would not be a safe per-example id.
|
|
84
|
+
"""
|
|
85
|
+
for key in ("example-id", "example_id"):
|
|
86
|
+
value = record.get(key)
|
|
87
|
+
if value is not None:
|
|
88
|
+
return str(value)
|
|
89
|
+
return f"folio:{line_no}"
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _example_from_record(record: dict, line_no: int,
|
|
93
|
+
known_bad_ids: FrozenSet[str]) -> DatasetExample:
|
|
94
|
+
premises = tuple(record.get("premises") or ())
|
|
95
|
+
premises_fol = tuple(record.get("premises-FOL") or ())
|
|
96
|
+
conclusion = record.get("conclusion")
|
|
97
|
+
conclusion_fol = record.get("conclusion-FOL")
|
|
98
|
+
label = record.get("label")
|
|
99
|
+
example_id = _resolve_id(record, line_no)
|
|
100
|
+
|
|
101
|
+
meta = {
|
|
102
|
+
k: v for k, v in record.items()
|
|
103
|
+
if k not in ("premises", "premises-FOL", "conclusion",
|
|
104
|
+
"conclusion-FOL", "label")
|
|
105
|
+
}
|
|
106
|
+
meta["line_no"] = line_no
|
|
107
|
+
|
|
108
|
+
return DatasetExample(
|
|
109
|
+
id=example_id,
|
|
110
|
+
nl_premises=premises,
|
|
111
|
+
fol_premises=premises_fol,
|
|
112
|
+
nl_conclusion=conclusion,
|
|
113
|
+
fol_conclusion=conclusion_fol,
|
|
114
|
+
label=label,
|
|
115
|
+
known_bad=example_id in known_bad_ids,
|
|
116
|
+
meta=meta,
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def load_folio(path: Union[str, Path], *,
|
|
121
|
+
known_bad_ids: FrozenSet[str] = frozenset()) -> Iterator[DatasetExample]:
|
|
122
|
+
"""Stream :class:`~unicode_logic_kit.eval.datasets.DatasetExample` from a
|
|
123
|
+
local FOLIO JSONL file.
|
|
124
|
+
|
|
125
|
+
Args:
|
|
126
|
+
path: path to a local ``.jsonl`` file in the verified FOLIO schema
|
|
127
|
+
(see module docstring) — one JSON object per non-blank line.
|
|
128
|
+
NEVER downloaded by this function; obtain the file from
|
|
129
|
+
https://github.com/Yale-LILY/FOLIO yourself.
|
|
130
|
+
known_bad_ids: ids (see :func:`_resolve_id`) whose gold annotation is
|
|
131
|
+
known to be broken (e.g. from a prior human review or an
|
|
132
|
+
:func:`~unicode_logic_kit.eval.datasets.audit_examples` pass on an
|
|
133
|
+
earlier load). Every yielded example with a matching id gets
|
|
134
|
+
``known_bad=True``; everything else gets ``known_bad=False``.
|
|
135
|
+
Defaults to an empty set (nothing pre-flagged).
|
|
136
|
+
|
|
137
|
+
Yields:
|
|
138
|
+
One :class:`~unicode_logic_kit.eval.datasets.DatasetExample` per
|
|
139
|
+
non-blank JSONL line, in file order. ``premises``/``premises-FOL``
|
|
140
|
+
(mapped to ``nl_premises``/``fol_premises``) are guaranteed to be the
|
|
141
|
+
same length only insofar as the source file guarantees it — this
|
|
142
|
+
loader does not enforce or re-check that pairing (a length mismatch
|
|
143
|
+
would show up as a defect during evaluation, not silently here).
|
|
144
|
+
|
|
145
|
+
Raises:
|
|
146
|
+
FileNotFoundError: ``path`` does not exist.
|
|
147
|
+
json.JSONDecodeError: a non-blank line is not valid JSON — this is
|
|
148
|
+
NOT swallowed; a malformed dataset file is a loud failure, not a
|
|
149
|
+
silently-skipped row.
|
|
150
|
+
"""
|
|
151
|
+
path = Path(path)
|
|
152
|
+
with path.open("r", encoding="utf-8") as fh:
|
|
153
|
+
for line_no, raw_line in enumerate(fh):
|
|
154
|
+
line = raw_line.strip()
|
|
155
|
+
if not line:
|
|
156
|
+
continue
|
|
157
|
+
record = json.loads(line)
|
|
158
|
+
yield _example_from_record(record, line_no, known_bad_ids)
|
|
@@ -0,0 +1,418 @@
|
|
|
1
|
+
"""Adapter for the FraCaS textual-inference problem set — pure NLI, no FOL.
|
|
2
|
+
|
|
3
|
+
Every other adapter in this package carries gold FOL (or generates it from a
|
|
4
|
+
structured source). FraCaS carries NONE, by construction: its problems are
|
|
5
|
+
natural-language premises plus a hypothesis and a three-valued answer, and
|
|
6
|
+
there is no logic field anywhere in its DTD. That is exactly why it earns a
|
|
7
|
+
place here — the kit's entailment verdict is three-valued too, so FraCaS's
|
|
8
|
+
``yes``/``no``/``unknown`` maps onto ``premises ⊨ h`` / ``premises ⊨ ¬h`` /
|
|
9
|
+
neither WITHOUT any interpretive glue, which makes it a reference target for
|
|
10
|
+
an NL→logic pipeline whose translation step lives OUTSIDE this library (see
|
|
11
|
+
:func:`solve_example`: the translation is an injected callable, never a
|
|
12
|
+
model this package calls).
|
|
13
|
+
|
|
14
|
+
Source and verified schema
|
|
15
|
+
---------------------------
|
|
16
|
+
Verified 2026-08-19 directly against the canonical machine-readable edition,
|
|
17
|
+
``https://nlp.stanford.edu/~wcmac/downloads/fracas.xml`` (XML conversion by
|
|
18
|
+
Bill MacCartney of the FraCaS Consortium's 1996 deliverable "Using the
|
|
19
|
+
Framework", Cooper et al.). Like every loader in this package it reads a
|
|
20
|
+
LOCAL file the caller already obtained — nothing here downloads anything.
|
|
21
|
+
|
|
22
|
+
The file's own header documents the representation; the numbers below are
|
|
23
|
+
this adapter's independent re-measurement of the file it parses:
|
|
24
|
+
|
|
25
|
+
* **346 problems**, ids ``"001"`` … ``"346"``, unique, zero-padded to three
|
|
26
|
+
digits (this adapter prefixes them: ``"fracas:001"``).
|
|
27
|
+
* **536 premises**, as ``<p idx="n">`` children — verified contiguous and
|
|
28
|
+
1-based in every problem (192 problems have one premise, 122 two, 29
|
|
29
|
+
three, 2 four, 1 five). Read in ``idx`` order, not document order.
|
|
30
|
+
* ``<q>`` the original question, ``<h>`` the declarative hypothesis, ``<a>``
|
|
31
|
+
the source document's answer text (``"Yes"``, ``"Don't know"``, but also
|
|
32
|
+
qualified phrases like ``"Not many"``), optional ``<why>`` (110) and
|
|
33
|
+
``<note>`` (33).
|
|
34
|
+
* ``fracas_answer`` ∈ ``yes`` (203) / ``unknown`` (98) / ``no`` (33) /
|
|
35
|
+
``undef`` (12) — the canonicalisation of ``<a>``; this is the ``label``.
|
|
36
|
+
* ``fracas_nonstandard="true"`` on the 41 problems whose ``<a>`` is not one
|
|
37
|
+
of the three canonical answers.
|
|
38
|
+
* Sections are NOT attributes: they are ``<comment class="section">`` /
|
|
39
|
+
``"subsection"`` / ``"subsubsection"`` markers between problems (9 / 47 /
|
|
40
|
+
10 of them), so section membership is DOCUMENT ORDER. This adapter tracks
|
|
41
|
+
them as it walks and resets the finer levels whenever a coarser one
|
|
42
|
+
changes — a problem can therefore never inherit a stale subsection from
|
|
43
|
+
the previous section.
|
|
44
|
+
|
|
45
|
+
Honest limitations
|
|
46
|
+
-------------------
|
|
47
|
+
* **Four problems (276, 305, 309, 310) have an EMPTY ``<q>`` and ``<h>``** —
|
|
48
|
+
the source document has no question for them. They load (nothing is
|
|
49
|
+
dropped silently) with ``nl_conclusion=None`` and are refused by
|
|
50
|
+
:func:`solve_example` with a named error rather than scored against an
|
|
51
|
+
absent hypothesis. All four are also ``undef``.
|
|
52
|
+
* **``undef`` is not a fourth answer class**, it marks a problem whose
|
|
53
|
+
source answer is not canonicalisable at all. Such examples load with
|
|
54
|
+
``label="undef"``; :func:`solve_example` will still PREDICT for the eight
|
|
55
|
+
of them that have a hypothesis (predicting is not scoring), and the caller
|
|
56
|
+
is expected to exclude them from any accuracy figure.
|
|
57
|
+
* ``fol_premises`` is always empty and ``fol_conclusion`` always ``None``:
|
|
58
|
+
there is no gold FOL to audit, so
|
|
59
|
+
:func:`~unicode_logic_kit.eval.datasets.audit_examples` reports these
|
|
60
|
+
examples as ``ok`` VACUOUSLY. That is not a claim about the data.
|
|
61
|
+
* The answer text in ``<a>`` is kept verbatim in ``meta["answer_text"]``,
|
|
62
|
+
including the qualified ones — canonicalising them further would be this
|
|
63
|
+
adapter inventing gold labels.
|
|
64
|
+
"""
|
|
65
|
+
|
|
66
|
+
import re
|
|
67
|
+
import xml.etree.ElementTree as ET
|
|
68
|
+
from pathlib import Path
|
|
69
|
+
from typing import (
|
|
70
|
+
Callable, Dict, FrozenSet, Iterable, Iterator, List, Optional, Union,
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
from ._base import DatasetExample, _register_dataset_info
|
|
74
|
+
|
|
75
|
+
__all__ = ["load_fracas", "solve_example", "ace_census", "FRACAS_ANSWERS"]
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
#: The canonical values of the ``fracas_answer`` attribute. ``undef`` is a
|
|
79
|
+
#: "no canonical answer exists" marker, not a fourth answer — see the module
|
|
80
|
+
#: docstring.
|
|
81
|
+
FRACAS_ANSWERS = ("yes", "no", "unknown", "undef")
|
|
82
|
+
|
|
83
|
+
_SECTION_LEVELS = ("section", "subsection", "subsubsection")
|
|
84
|
+
_HEADING_RE = re.compile(r"^([\d.]+)\s+(.*)$", re.DOTALL)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
_register_dataset_info(
|
|
88
|
+
"fracas",
|
|
89
|
+
license=("no explicit licence statement in the source file; the XML "
|
|
90
|
+
"edition asks for credit for the conversion, and the problems "
|
|
91
|
+
"derive from the FraCaS Consortium's 1996 deliverable"),
|
|
92
|
+
source_url="https://nlp.stanford.edu/~wcmac/downloads/fracas.xml",
|
|
93
|
+
citation_hint=('FraCaS Consortium (Cooper et al.), "Using the '
|
|
94
|
+
'Framework", 1996; XML edition by Bill MacCartney.'),
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
# ---------------------------------------------------------------------------
|
|
99
|
+
# Reading
|
|
100
|
+
# ---------------------------------------------------------------------------
|
|
101
|
+
|
|
102
|
+
def _text(element: Optional[ET.Element]) -> Optional[str]:
|
|
103
|
+
"""Element text with XML indentation collapsed — ``None`` when absent or
|
|
104
|
+
empty (the four question-less problems), never the empty string."""
|
|
105
|
+
if element is None or element.text is None:
|
|
106
|
+
return None
|
|
107
|
+
collapsed = " ".join(element.text.split())
|
|
108
|
+
return collapsed or None
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _heading(raw: Optional[str]) -> Dict[str, Optional[str]]:
|
|
112
|
+
"""``"1.2 Monotonicity (…)"`` → number and title, kept separate so a
|
|
113
|
+
caller filters on the STABLE number rather than on prose."""
|
|
114
|
+
if raw is None:
|
|
115
|
+
return {"number": None, "title": None}
|
|
116
|
+
match = _HEADING_RE.match(raw)
|
|
117
|
+
if match is None:
|
|
118
|
+
return {"number": None, "title": raw}
|
|
119
|
+
return {"number": match.group(1), "title": " ".join(match.group(2).split())}
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _premises(problem: ET.Element) -> List[str]:
|
|
123
|
+
"""The ``<p>`` texts in ``idx`` order, with the ordering CHECKED: the
|
|
124
|
+
file's indices are contiguous and 1-based throughout, so anything else
|
|
125
|
+
is a corrupted input and says so instead of being silently reordered."""
|
|
126
|
+
numbered = []
|
|
127
|
+
for element in problem.findall("p"):
|
|
128
|
+
raw_idx = element.get("idx")
|
|
129
|
+
if raw_idx is None or not raw_idx.isdigit():
|
|
130
|
+
raise ValueError(
|
|
131
|
+
f"fracas: problem {problem.get('id')!r} has a <p> without a "
|
|
132
|
+
f"numeric idx (got {raw_idx!r})")
|
|
133
|
+
text = _text(element)
|
|
134
|
+
if text is None:
|
|
135
|
+
raise ValueError(
|
|
136
|
+
f"fracas: problem {problem.get('id')!r} has an empty premise "
|
|
137
|
+
f"at idx {raw_idx}")
|
|
138
|
+
numbered.append((int(raw_idx), text))
|
|
139
|
+
numbered.sort()
|
|
140
|
+
if [i for i, _ in numbered] != list(range(1, len(numbered) + 1)):
|
|
141
|
+
raise ValueError(
|
|
142
|
+
f"fracas: problem {problem.get('id')!r} has non-contiguous "
|
|
143
|
+
f"premise indices {[i for i, _ in numbered]}")
|
|
144
|
+
return [text for _, text in numbered]
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def load_fracas(path: Union[str, Path], *,
|
|
148
|
+
sections: Optional[Iterable[str]] = None,
|
|
149
|
+
answers: Optional[Iterable[str]] = None,
|
|
150
|
+
known_bad_ids: FrozenSet[str] = frozenset(),
|
|
151
|
+
) -> Iterator[DatasetExample]:
|
|
152
|
+
"""Read the FraCaS XML into :class:`DatasetExample` objects, in file order.
|
|
153
|
+
|
|
154
|
+
Field mapping (see the module docstring for what each source element is):
|
|
155
|
+
``nl_premises`` = the ``<p>`` texts in ``idx`` order, ``nl_conclusion`` =
|
|
156
|
+
``<h>`` (``None`` for the four question-less problems), ``label`` =
|
|
157
|
+
``fracas_answer``, and ``fol_premises``/``fol_conclusion`` stay empty —
|
|
158
|
+
FraCaS has no logic annotation. Everything else from the record survives
|
|
159
|
+
in ``meta``: ``question``, ``answer_text``, ``why``, ``note``,
|
|
160
|
+
``nonstandard``, ``premise_count``, and the section / subsection /
|
|
161
|
+
subsubsection numbers and titles.
|
|
162
|
+
|
|
163
|
+
Args:
|
|
164
|
+
path: the local ``fracas.xml``.
|
|
165
|
+
sections: keep only problems in these SECTION NUMBERS (``{"1", "3"}``
|
|
166
|
+
— the stable identifier, matched against the top-level section,
|
|
167
|
+
so ``"1"`` keeps all of ``1.x``). ``None`` keeps everything.
|
|
168
|
+
answers: keep only these ``fracas_answer`` values (e.g.
|
|
169
|
+
``{"yes", "no", "unknown"}`` to drop the twelve ``undef``
|
|
170
|
+
problems). ``None`` keeps everything, ``undef`` included — this
|
|
171
|
+
loader never drops them on its own.
|
|
172
|
+
known_bad_ids: ids (in the prefixed ``"fracas:001"`` form) to flag as
|
|
173
|
+
``known_bad``; the same caller-curated mechanic every adapter has.
|
|
174
|
+
|
|
175
|
+
Raises:
|
|
176
|
+
ValueError: the file is not a FraCaS problem set, a problem lacks its
|
|
177
|
+
id or ``fracas_answer``, an answer is outside
|
|
178
|
+
:data:`FRACAS_ANSWERS`, ids repeat, or premise indices are not
|
|
179
|
+
contiguous — a malformed input is named, never worked around.
|
|
180
|
+
"""
|
|
181
|
+
wanted_sections = None if sections is None else {str(s) for s in sections}
|
|
182
|
+
wanted_answers = None if answers is None else {str(a) for a in answers}
|
|
183
|
+
if wanted_answers is not None:
|
|
184
|
+
unknown = wanted_answers - set(FRACAS_ANSWERS)
|
|
185
|
+
if unknown:
|
|
186
|
+
raise ValueError(
|
|
187
|
+
f"fracas: answers={sorted(unknown)} is outside "
|
|
188
|
+
f"{list(FRACAS_ANSWERS)}")
|
|
189
|
+
|
|
190
|
+
root = ET.parse(str(path)).getroot()
|
|
191
|
+
if root.tag != "fracas-problems":
|
|
192
|
+
raise ValueError(
|
|
193
|
+
f"fracas: {path} has root element {root.tag!r}, expected "
|
|
194
|
+
"'fracas-problems' — is this the FraCaS XML?")
|
|
195
|
+
|
|
196
|
+
headings: Dict[str, Dict[str, Optional[str]]] = {
|
|
197
|
+
level: _heading(None) for level in _SECTION_LEVELS}
|
|
198
|
+
seen = set()
|
|
199
|
+
|
|
200
|
+
for element in root:
|
|
201
|
+
if element.tag == "comment":
|
|
202
|
+
level = element.get("class")
|
|
203
|
+
if level in _SECTION_LEVELS:
|
|
204
|
+
headings[level] = _heading(_text(element))
|
|
205
|
+
# A coarser heading invalidates every finer one, so a
|
|
206
|
+
# problem can never inherit a stale subsection.
|
|
207
|
+
for finer in _SECTION_LEVELS[_SECTION_LEVELS.index(level) + 1:]:
|
|
208
|
+
headings[finer] = _heading(None)
|
|
209
|
+
continue
|
|
210
|
+
if element.tag != "problem":
|
|
211
|
+
continue
|
|
212
|
+
|
|
213
|
+
raw_id = element.get("id")
|
|
214
|
+
if not raw_id:
|
|
215
|
+
raise ValueError("fracas: a <problem> element has no id")
|
|
216
|
+
example_id = f"fracas:{raw_id}"
|
|
217
|
+
if example_id in seen:
|
|
218
|
+
raise ValueError(f"fracas: duplicate problem id {raw_id!r}")
|
|
219
|
+
seen.add(example_id)
|
|
220
|
+
|
|
221
|
+
answer = element.get("fracas_answer")
|
|
222
|
+
if answer is None:
|
|
223
|
+
raise ValueError(
|
|
224
|
+
f"fracas: problem {raw_id!r} has no fracas_answer attribute")
|
|
225
|
+
if answer not in FRACAS_ANSWERS:
|
|
226
|
+
raise ValueError(
|
|
227
|
+
f"fracas: problem {raw_id!r} has fracas_answer {answer!r}, "
|
|
228
|
+
f"outside {list(FRACAS_ANSWERS)}")
|
|
229
|
+
|
|
230
|
+
if (wanted_sections is not None
|
|
231
|
+
and headings["section"]["number"] not in wanted_sections):
|
|
232
|
+
continue
|
|
233
|
+
if wanted_answers is not None and answer not in wanted_answers:
|
|
234
|
+
continue
|
|
235
|
+
|
|
236
|
+
premises = _premises(element)
|
|
237
|
+
meta = {
|
|
238
|
+
"question": _text(element.find("q")),
|
|
239
|
+
"answer_text": _text(element.find("a")),
|
|
240
|
+
"why": _text(element.find("why")),
|
|
241
|
+
"note": _text(element.find("note")),
|
|
242
|
+
"nonstandard": element.get("fracas_nonstandard") == "true",
|
|
243
|
+
"premise_count": len(premises),
|
|
244
|
+
}
|
|
245
|
+
for level in _SECTION_LEVELS:
|
|
246
|
+
meta[level] = headings[level]["number"]
|
|
247
|
+
meta[f"{level}_title"] = headings[level]["title"]
|
|
248
|
+
|
|
249
|
+
yield DatasetExample(
|
|
250
|
+
id=example_id,
|
|
251
|
+
nl_premises=tuple(premises),
|
|
252
|
+
fol_premises=(),
|
|
253
|
+
nl_conclusion=_text(element.find("h")),
|
|
254
|
+
fol_conclusion=None,
|
|
255
|
+
label=answer,
|
|
256
|
+
known_bad=example_id in known_bad_ids,
|
|
257
|
+
meta=meta,
|
|
258
|
+
)
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
# ---------------------------------------------------------------------------
|
|
262
|
+
# Deciding — with the translation injected by the caller
|
|
263
|
+
# ---------------------------------------------------------------------------
|
|
264
|
+
|
|
265
|
+
def solve_example(example: DatasetExample, *,
|
|
266
|
+
translate: Callable[[str], object],
|
|
267
|
+
on_indefinite: str = "label", **prove_kwargs) -> dict:
|
|
268
|
+
"""Decide one FraCaS problem end-to-end — the translation is YOURS.
|
|
269
|
+
|
|
270
|
+
FraCaS ships no formulas, so this helper takes ``translate``: a callable
|
|
271
|
+
mapping one natural-language sentence to either a formula string (parsed
|
|
272
|
+
with :func:`unicode_logic_kit.api.parse_any`) or an already-built kit node.
|
|
273
|
+
That is the seam where an external system — a semantic parser, a
|
|
274
|
+
hand-written table, a language model driven by the caller — plugs in;
|
|
275
|
+
this package deliberately calls no such system itself.
|
|
276
|
+
|
|
277
|
+
The rest is the same three-valued cascade the other adapters use, and it
|
|
278
|
+
matches FraCaS's own answer semantics exactly: ``"yes"`` iff premises ⊨
|
|
279
|
+
hypothesis, ``"no"`` iff premises ⊨ ¬hypothesis, ``"unknown"`` otherwise.
|
|
280
|
+
Extra ``prove_kwargs`` reach :func:`unicode_logic_kit.api.prove` verbatim,
|
|
281
|
+
so the prover is the caller's choice.
|
|
282
|
+
|
|
283
|
+
``on_indefinite`` handles a NON-DEFINITIVE prover outcome (timeout, hit
|
|
284
|
+
bound, honest incompleteness) when neither direction was proved:
|
|
285
|
+
|
|
286
|
+
- ``"label"`` (default): predict ``"unknown"``.
|
|
287
|
+
- ``"abstain"``: ``"unknown"`` only when BOTH directions came back
|
|
288
|
+
definitively refuted (underdetermination established by countermodels);
|
|
289
|
+
any indefinite leg yields ``predicted=None``, so a timeout can never be
|
|
290
|
+
scored as a correct "unknown".
|
|
291
|
+
- ``"raise"``: like ``"abstain"`` but raises instead.
|
|
292
|
+
|
|
293
|
+
Returns a dict with ``predicted`` (``"yes"``/``"no"``/``"unknown"``/
|
|
294
|
+
``None``), ``label`` (the gold answer, ``"undef"`` included — scoring
|
|
295
|
+
against it is the caller's decision), ``verdict``/``verdict_negated``
|
|
296
|
+
(verdict dicts; the negated one is ``None`` when the positive direction
|
|
297
|
+
already settled it), and the translated ``premises``/``hypothesis`` in
|
|
298
|
+
kit notation, so a wrong prediction can be traced back to the
|
|
299
|
+
translation that caused it.
|
|
300
|
+
|
|
301
|
+
Raises:
|
|
302
|
+
ValueError: the example has no hypothesis (the four question-less
|
|
303
|
+
problems), ``on_indefinite`` is not one of the three modes, or a
|
|
304
|
+
translated string does not parse.
|
|
305
|
+
"""
|
|
306
|
+
from ... import api
|
|
307
|
+
from ...fol.nodes import Node, Not
|
|
308
|
+
|
|
309
|
+
if on_indefinite not in ("label", "abstain", "raise"):
|
|
310
|
+
raise ValueError(
|
|
311
|
+
f"fracas: on_indefinite must be 'label', 'abstain' or 'raise', "
|
|
312
|
+
f"got {on_indefinite!r}")
|
|
313
|
+
if example.nl_conclusion is None:
|
|
314
|
+
raise ValueError(
|
|
315
|
+
f"fracas: example {example.id} has no hypothesis (the source "
|
|
316
|
+
"document has no question for it) — nothing to decide.")
|
|
317
|
+
|
|
318
|
+
def _formula(sentence: str) -> "Node":
|
|
319
|
+
produced = translate(sentence)
|
|
320
|
+
if isinstance(produced, Node):
|
|
321
|
+
return produced
|
|
322
|
+
if not isinstance(produced, str):
|
|
323
|
+
raise ValueError(
|
|
324
|
+
f"fracas: example {example.id}: translate({sentence!r}) "
|
|
325
|
+
f"returned {type(produced).__name__}, expected a formula "
|
|
326
|
+
"string or a kit node")
|
|
327
|
+
parsed = api.parse_any(produced)
|
|
328
|
+
if not parsed.ok:
|
|
329
|
+
raise ValueError(
|
|
330
|
+
f"fracas: example {example.id}: the translation "
|
|
331
|
+
f"{produced!r} of {sentence!r} does not parse")
|
|
332
|
+
return parsed.formula
|
|
333
|
+
|
|
334
|
+
premises = [_formula(sentence) for sentence in example.nl_premises]
|
|
335
|
+
hypothesis = _formula(example.nl_conclusion)
|
|
336
|
+
|
|
337
|
+
result = {
|
|
338
|
+
"label": example.label,
|
|
339
|
+
"premises": [p.to_unicode_str() for p in premises],
|
|
340
|
+
"hypothesis": hypothesis.to_unicode_str(),
|
|
341
|
+
}
|
|
342
|
+
verdict = api.prove(hypothesis, premises, **prove_kwargs)
|
|
343
|
+
if verdict.status == "proved":
|
|
344
|
+
result.update(predicted="yes", verdict=verdict.to_dict(),
|
|
345
|
+
verdict_negated=None)
|
|
346
|
+
return result
|
|
347
|
+
|
|
348
|
+
negated = api.prove(Not(hypothesis), premises, **prove_kwargs)
|
|
349
|
+
if negated.status == "proved":
|
|
350
|
+
predicted: Optional[str] = "no"
|
|
351
|
+
elif on_indefinite == "label":
|
|
352
|
+
predicted = "unknown"
|
|
353
|
+
elif verdict.status == "refuted" and negated.status == "refuted":
|
|
354
|
+
# Underdetermination ESTABLISHED both ways: "unknown" is a definitive
|
|
355
|
+
# answer here, so even abstain/raise report it.
|
|
356
|
+
predicted = "unknown"
|
|
357
|
+
elif on_indefinite == "raise":
|
|
358
|
+
raise ValueError(
|
|
359
|
+
f"fracas: example {example.id}: indefinite prover outcome "
|
|
360
|
+
f"(goal: {verdict.status}/{verdict.reason}, negated: "
|
|
361
|
+
f"{negated.status}/{negated.reason}) with on_indefinite='raise'.")
|
|
362
|
+
else: # "abstain"
|
|
363
|
+
predicted = None
|
|
364
|
+
|
|
365
|
+
result.update(predicted=predicted, verdict=verdict.to_dict(),
|
|
366
|
+
verdict_negated=negated.to_dict())
|
|
367
|
+
return result
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
# ---------------------------------------------------------------------------
|
|
371
|
+
# How much of it is controlled English?
|
|
372
|
+
# ---------------------------------------------------------------------------
|
|
373
|
+
|
|
374
|
+
def ace_census(examples: Iterable[DatasetExample], *,
|
|
375
|
+
ulex: Optional[str] = None, timeout: float = 30.0,
|
|
376
|
+
) -> List[dict]:
|
|
377
|
+
"""Per-SENTENCE report: which FraCaS sentences does APE accept as ACE?
|
|
378
|
+
|
|
379
|
+
A measurement, not a score. FraCaS is short, deliberately plain English,
|
|
380
|
+
so it is the natural corpus for asking how far Attempto Controlled
|
|
381
|
+
English reaches as a target notation — and the answer comes per
|
|
382
|
+
sentence, with APE's own diagnosis attached, never as a single aggregate
|
|
383
|
+
this function decides for you (group the rows by ``section`` yourself).
|
|
384
|
+
|
|
385
|
+
Needs a reachable APE binary
|
|
386
|
+
(:func:`unicode_logic_kit.ace.ape_available`); ``ulex`` is passed through
|
|
387
|
+
as APE's user lexicon, which matters because APE's built-in lexicon is
|
|
388
|
+
small and a missing word is reported as "not ACE" like any other
|
|
389
|
+
refusal.
|
|
390
|
+
|
|
391
|
+
Returns one dict per sentence, in example order: ``id``, ``section``,
|
|
392
|
+
``role`` (``"premise"``/``"hypothesis"``), ``index`` (position within the
|
|
393
|
+
premises, ``None`` for the hypothesis), ``sentence``, ``status``
|
|
394
|
+
(:class:`~unicode_logic_kit.ace.runner.CoverageRow`'s vocabulary:
|
|
395
|
+
``ok``/``tptp_unsupported``/``tptp_unread``/``not_ace``/``infra``) and
|
|
396
|
+
``detail``.
|
|
397
|
+
"""
|
|
398
|
+
from ...ace import ace_coverage
|
|
399
|
+
|
|
400
|
+
rows: List[dict] = []
|
|
401
|
+
for example in examples:
|
|
402
|
+
sentences = [("premise", i, s)
|
|
403
|
+
for i, s in enumerate(example.nl_premises)]
|
|
404
|
+
if example.nl_conclusion is not None:
|
|
405
|
+
sentences.append(("hypothesis", None, example.nl_conclusion))
|
|
406
|
+
coverage = ace_coverage([s for _, _, s in sentences],
|
|
407
|
+
ulex=ulex, timeout=timeout)
|
|
408
|
+
for (role, index, sentence), row in zip(sentences, coverage):
|
|
409
|
+
rows.append({
|
|
410
|
+
"id": example.id,
|
|
411
|
+
"section": example.meta.get("section"),
|
|
412
|
+
"role": role,
|
|
413
|
+
"index": index,
|
|
414
|
+
"sentence": sentence,
|
|
415
|
+
"status": row.status,
|
|
416
|
+
"detail": row.detail,
|
|
417
|
+
})
|
|
418
|
+
return rows
|