unicode-logic-kit 0.31.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. unicode_logic_kit/__init__.py +385 -0
  2. unicode_logic_kit/__main__.py +520 -0
  3. unicode_logic_kit/_deadline.py +219 -0
  4. unicode_logic_kit/ace/__init__.py +126 -0
  5. unicode_logic_kit/ace/_align.py +135 -0
  6. unicode_logic_kit/ace/chem_lexicon.py +128 -0
  7. unicode_logic_kit/ace/drs_reader.py +570 -0
  8. unicode_logic_kit/ace/mapping.py +666 -0
  9. unicode_logic_kit/ace/reverse_modal.py +138 -0
  10. unicode_logic_kit/ace/runner.py +551 -0
  11. unicode_logic_kit/ace/translate.py +452 -0
  12. unicode_logic_kit/ace/verbalize.py +1070 -0
  13. unicode_logic_kit/api.py +1284 -0
  14. unicode_logic_kit/atp/__init__.py +177 -0
  15. unicode_logic_kit/atp/_ascii_names.py +113 -0
  16. unicode_logic_kit/atp/_html.py +72 -0
  17. unicode_logic_kit/atp/_substructural_input.py +228 -0
  18. unicode_logic_kit/atp/_tff_problem.py +715 -0
  19. unicode_logic_kit/atp/_tptp_problem.py +1111 -0
  20. unicode_logic_kit/atp/_writer_support.py +289 -0
  21. unicode_logic_kit/atp/clingo_backend.py +1180 -0
  22. unicode_logic_kit/atp/cvc5_backend.py +1385 -0
  23. unicode_logic_kit/atp/eprover_backend.py +732 -0
  24. unicode_logic_kit/atp/finite_domain.py +1055 -0
  25. unicode_logic_kit/atp/fitch.py +1547 -0
  26. unicode_logic_kit/atp/fitch_search.py +551 -0
  27. unicode_logic_kit/atp/hets_backend.py +339 -0
  28. unicode_logic_kit/atp/hybrid_down.py +120 -0
  29. unicode_logic_kit/atp/incremental.py +250 -0
  30. unicode_logic_kit/atp/kripke_enum.py +741 -0
  31. unicode_logic_kit/atp/lambek.py +436 -0
  32. unicode_logic_kit/atp/leo3_backend.py +332 -0
  33. unicode_logic_kit/atp/linear.py +738 -0
  34. unicode_logic_kit/atp/lj.py +705 -0
  35. unicode_logic_kit/atp/logic_backends.py +566 -0
  36. unicode_logic_kit/atp/ltl_tableau.py +1084 -0
  37. unicode_logic_kit/atp/minizinc_backend.py +1402 -0
  38. unicode_logic_kit/atp/modal_tableau.py +1382 -0
  39. unicode_logic_kit/atp/nanocop_backend.py +410 -0
  40. unicode_logic_kit/atp/portfolio.py +489 -0
  41. unicode_logic_kit/atp/protocol.py +1803 -0
  42. unicode_logic_kit/atp/prover9_entailment.py +1153 -0
  43. unicode_logic_kit/atp/resolution.py +1376 -0
  44. unicode_logic_kit/atp/resolution_check.py +1114 -0
  45. unicode_logic_kit/atp/sequent.py +1050 -0
  46. unicode_logic_kit/atp/tableau.py +921 -0
  47. unicode_logic_kit/atp/tableau_check.py +543 -0
  48. unicode_logic_kit/atp/tptp_ncl.py +811 -0
  49. unicode_logic_kit/atp/tptp_tff.py +1546 -0
  50. unicode_logic_kit/atp/tstp.py +1333 -0
  51. unicode_logic_kit/atp/tstp_check.py +1096 -0
  52. unicode_logic_kit/atp/twee_backend.py +236 -0
  53. unicode_logic_kit/atp/twee_check.py +711 -0
  54. unicode_logic_kit/atp/twee_entailment.py +953 -0
  55. unicode_logic_kit/atp/vampire_entailment.py +540 -0
  56. unicode_logic_kit/atp/z3_arith.py +470 -0
  57. unicode_logic_kit/atp/z3_equivalence.py +36 -0
  58. unicode_logic_kit/atp/z3_fuzzy.py +362 -0
  59. unicode_logic_kit/atp/z3_input.py +500 -0
  60. unicode_logic_kit/atp/z3_models.py +208 -0
  61. unicode_logic_kit/chem/__init__.py +88 -0
  62. unicode_logic_kit/chem/_naming.py +284 -0
  63. unicode_logic_kit/chem/cache.py +185 -0
  64. unicode_logic_kit/chem/interop.py +244 -0
  65. unicode_logic_kit/chem/mol.py +525 -0
  66. unicode_logic_kit/chem/signature.py +112 -0
  67. unicode_logic_kit/comorphism.py +497 -0
  68. unicode_logic_kit/dl/__init__.py +384 -0
  69. unicode_logic_kit/dl/classification.py +227 -0
  70. unicode_logic_kit/dl/concepts.py +632 -0
  71. unicode_logic_kit/dl/datatypes.py +818 -0
  72. unicode_logic_kit/dl/owl_functional.py +2433 -0
  73. unicode_logic_kit/dl/owl_manchester.py +1637 -0
  74. unicode_logic_kit/dl/owl_reasoner.py +790 -0
  75. unicode_logic_kit/dl/parser.py +391 -0
  76. unicode_logic_kit/dl/tableau.py +4048 -0
  77. unicode_logic_kit/dl/translate.py +2704 -0
  78. unicode_logic_kit/drt/__init__.py +94 -0
  79. unicode_logic_kit/drt/export.py +179 -0
  80. unicode_logic_kit/drt/nodes.py +506 -0
  81. unicode_logic_kit/drt/parser.py +965 -0
  82. unicode_logic_kit/drt/resolve.py +195 -0
  83. unicode_logic_kit/drt/reverse.py +175 -0
  84. unicode_logic_kit/eval/__init__.py +106 -0
  85. unicode_logic_kit/eval/batch.py +382 -0
  86. unicode_logic_kit/eval/canonical.py +663 -0
  87. unicode_logic_kit/eval/chem_batch.py +606 -0
  88. unicode_logic_kit/eval/converses.py +200 -0
  89. unicode_logic_kit/eval/datasets/__init__.py +136 -0
  90. unicode_logic_kit/eval/datasets/_base.py +263 -0
  91. unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
  92. unicode_logic_kit/eval/datasets/c3po.py +678 -0
  93. unicode_logic_kit/eval/datasets/folio.py +158 -0
  94. unicode_logic_kit/eval/datasets/fracas.py +418 -0
  95. unicode_logic_kit/eval/datasets/groves.py +191 -0
  96. unicode_logic_kit/eval/datasets/logicbench.py +467 -0
  97. unicode_logic_kit/eval/datasets/logicnli.py +303 -0
  98. unicode_logic_kit/eval/datasets/malls.py +133 -0
  99. unicode_logic_kit/eval/datasets/pfolio.py +594 -0
  100. unicode_logic_kit/eval/datasets/pmb.py +242 -0
  101. unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
  102. unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
  103. unicode_logic_kit/eval/datasets/proverqa.py +674 -0
  104. unicode_logic_kit/eval/datasets/willow.py +478 -0
  105. unicode_logic_kit/eval/equivalence.py +466 -0
  106. unicode_logic_kit/eval/exercise_gen.py +533 -0
  107. unicode_logic_kit/eval/explain.py +791 -0
  108. unicode_logic_kit/eval/generality.py +750 -0
  109. unicode_logic_kit/eval/metric_hf.py +458 -0
  110. unicode_logic_kit/eval/predicate_match.py +343 -0
  111. unicode_logic_kit/eval/theory_check.py +1170 -0
  112. unicode_logic_kit/eval/validate.py +306 -0
  113. unicode_logic_kit/fol/__init__.py +177 -0
  114. unicode_logic_kit/fol/_atom_keys.py +510 -0
  115. unicode_logic_kit/fol/_fol_nodes.py +3586 -0
  116. unicode_logic_kit/fol/_free_parameters.py +105 -0
  117. unicode_logic_kit/fol/_ho_nodes.py +448 -0
  118. unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
  119. unicode_logic_kit/fol/_identifiers.py +1091 -0
  120. unicode_logic_kit/fol/_lambek_nodes.py +112 -0
  121. unicode_logic_kit/fol/_linear_nodes.py +352 -0
  122. unicode_logic_kit/fol/_modal_nodes.py +1467 -0
  123. unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
  124. unicode_logic_kit/fol/_numeral_symbols.py +231 -0
  125. unicode_logic_kit/fol/_so_nodes.py +200 -0
  126. unicode_logic_kit/fol/_symbol_names.py +81 -0
  127. unicode_logic_kit/fol/_team_nodes.py +181 -0
  128. unicode_logic_kit/fol/_tptp_symbols.py +551 -0
  129. unicode_logic_kit/fol/_truth_constants.py +117 -0
  130. unicode_logic_kit/fol/casl_export.py +1135 -0
  131. unicode_logic_kit/fol/casl_import.py +929 -0
  132. unicode_logic_kit/fol/derivation.py +367 -0
  133. unicode_logic_kit/fol/dialect_detect.py +70 -0
  134. unicode_logic_kit/fol/dialect_repair.py +537 -0
  135. unicode_logic_kit/fol/frames.py +637 -0
  136. unicode_logic_kit/fol/grammars/terminals.lark +31 -0
  137. unicode_logic_kit/fol/lambda_tools.py +297 -0
  138. unicode_logic_kit/fol/latex_input.py +429 -0
  139. unicode_logic_kit/fol/modal_translation.py +944 -0
  140. unicode_logic_kit/fol/msflparser.py +1033 -0
  141. unicode_logic_kit/fol/naming.py +422 -0
  142. unicode_logic_kit/fol/nodes.py +241 -0
  143. unicode_logic_kit/fol/normalforms.py +492 -0
  144. unicode_logic_kit/fol/pal.py +287 -0
  145. unicode_logic_kit/fol/prolog_export.py +566 -0
  146. unicode_logic_kit/fol/prolog_input.py +505 -0
  147. unicode_logic_kit/fol/prover9_input.py +1325 -0
  148. unicode_logic_kit/fol/qml.py +1760 -0
  149. unicode_logic_kit/fol/qmltp_input.py +525 -0
  150. unicode_logic_kit/fol/sanitize.py +221 -0
  151. unicode_logic_kit/fol/serialize.py +79 -0
  152. unicode_logic_kit/fol/signature.py +1290 -0
  153. unicode_logic_kit/fol/simplify_check.py +544 -0
  154. unicode_logic_kit/fol/spans.py +594 -0
  155. unicode_logic_kit/fol/tptp_input.py +1503 -0
  156. unicode_logic_kit/fol/tptp_repair.py +941 -0
  157. unicode_logic_kit/fol/unification.py +157 -0
  158. unicode_logic_kit/fol/verbalize.py +263 -0
  159. unicode_logic_kit/hets/__init__.py +163 -0
  160. unicode_logic_kit/hets/bridge.py +142 -0
  161. unicode_logic_kit/hets/client.py +748 -0
  162. unicode_logic_kit/hets/docker.py +420 -0
  163. unicode_logic_kit/hets/dol.py +712 -0
  164. unicode_logic_kit/hets/haskell_json.py +355 -0
  165. unicode_logic_kit/hets/owl_backend.py +794 -0
  166. unicode_logic_kit/hets/owl_cli.py +598 -0
  167. unicode_logic_kit/hets/symbols.py +512 -0
  168. unicode_logic_kit/hol/__init__.py +140 -0
  169. unicode_logic_kit/hol/_ho_common.py +323 -0
  170. unicode_logic_kit/hol/_isabelle_binders.py +125 -0
  171. unicode_logic_kit/hol/classical.py +812 -0
  172. unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
  173. unicode_logic_kit/hol/deepshallow/_common.py +177 -0
  174. unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
  175. unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
  176. unicode_logic_kit/hol/deepshallow/modal.py +217 -0
  177. unicode_logic_kit/hol/deepshallow/qml.py +406 -0
  178. unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
  179. unicode_logic_kit/hol/free.py +753 -0
  180. unicode_logic_kit/hol/goedel.py +336 -0
  181. unicode_logic_kit/hol/ho_modal.py +1743 -0
  182. unicode_logic_kit/hol/intuitionistic.py +403 -0
  183. unicode_logic_kit/hol/isabelle_conditional.py +593 -0
  184. unicode_logic_kit/hol/isabelle_modal.py +1908 -0
  185. unicode_logic_kit/hol/isabelle_relevant.py +412 -0
  186. unicode_logic_kit/hol/isabelle_runner.py +1147 -0
  187. unicode_logic_kit/hol/isabelle_substructural.py +884 -0
  188. unicode_logic_kit/hol/lean.py +1018 -0
  189. unicode_logic_kit/hol/manyvalued.py +921 -0
  190. unicode_logic_kit/hol/secondorder.py +687 -0
  191. unicode_logic_kit/hol/thf_modal.py +941 -0
  192. unicode_logic_kit/hol/thirdorder.py +397 -0
  193. unicode_logic_kit/ilp/__init__.py +89 -0
  194. unicode_logic_kit/ilp/readback.py +389 -0
  195. unicode_logic_kit/ilp/separation.py +153 -0
  196. unicode_logic_kit/ilp/task.py +730 -0
  197. unicode_logic_kit/logic.py +163 -0
  198. unicode_logic_kit/mcp/__init__.py +28 -0
  199. unicode_logic_kit/mcp/__main__.py +5 -0
  200. unicode_logic_kit/mcp/chem_tools.py +1031 -0
  201. unicode_logic_kit/mcp/server.py +2453 -0
  202. unicode_logic_kit/mcp/syntax_spec.py +681 -0
  203. unicode_logic_kit/prob/__init__.py +53 -0
  204. unicode_logic_kit/prob/_bdd.py +225 -0
  205. unicode_logic_kit/prob/_column_gen.py +668 -0
  206. unicode_logic_kit/prob/distribution.py +686 -0
  207. unicode_logic_kit/prob/nilsson.py +470 -0
  208. unicode_logic_kit/py.typed +0 -0
  209. unicode_logic_kit/semantics/__init__.py +137 -0
  210. unicode_logic_kit/semantics/_modal_reject.py +156 -0
  211. unicode_logic_kit/semantics/action_models.py +466 -0
  212. unicode_logic_kit/semantics/asp_models.py +1200 -0
  213. unicode_logic_kit/semantics/conditional.py +580 -0
  214. unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
  215. unicode_logic_kit/semantics/free_logic.py +913 -0
  216. unicode_logic_kit/semantics/fuzzy.py +384 -0
  217. unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
  218. unicode_logic_kit/semantics/intuitionistic.py +581 -0
  219. unicode_logic_kit/semantics/kripke.py +1139 -0
  220. unicode_logic_kit/semantics/manyvalued.py +580 -0
  221. unicode_logic_kit/semantics/matrix.py +342 -0
  222. unicode_logic_kit/semantics/model_eval.py +1135 -0
  223. unicode_logic_kit/semantics/modelfinder.py +1036 -0
  224. unicode_logic_kit/semantics/nonmonotonic.py +372 -0
  225. unicode_logic_kit/semantics/relevant.py +331 -0
  226. unicode_logic_kit/semantics/secondorder.py +657 -0
  227. unicode_logic_kit/semantics/structures.py +352 -0
  228. unicode_logic_kit/semantics/tarski.py +975 -0
  229. unicode_logic_kit/semantics/team.py +315 -0
  230. unicode_logic_kit/semantics/team_translation.py +416 -0
  231. unicode_logic_kit/semantics/thirdorder.py +358 -0
  232. unicode_logic_kit/semantics/tnorm.py +85 -0
  233. unicode_logic_kit/semantics/truthtable.py +201 -0
  234. unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
  235. unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
  236. unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
  237. unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,158 @@
1
+ """Adapter for the FOLIO dataset (Han et al., "FOLIO: Natural Language
2
+ Reasoning with First-Order Logic") — local JSONL only, no network access.
3
+
4
+ Source and verified schema
5
+ ---------------------------
6
+ Canonical, unauthenticated source: https://github.com/Yale-LILY/FOLIO
7
+ (``data/v0.0/folio-train.jsonl`` / ``data/v0.0/folio-validation.jsonl``). A
8
+ mirror also exists at https://huggingface.co/datasets/yale-nlp/FOLIO, but
9
+ that mirror is access-gated (requires a Hugging Face login to download), so
10
+ it was NOT used to verify the schema below — everything here was verified
11
+ directly against the raw GitHub JSONL content.
12
+
13
+ Each verified ``v0.0`` JSONL line is a flat JSON object with exactly these
14
+ keys (confirmed by fetching and inspecting real rows of
15
+ ``folio-validation.jsonl``):
16
+
17
+ * ``"premises"`` — ``list[str]``, the natural-language premises.
18
+ * ``"premises-FOL"`` — ``list[str]``, the SAME LENGTH and ORDER as
19
+ ``"premises"``; each entry is that premise's gold FOL annotation, written
20
+ in this kit's own unicode surface syntax (``∀``/``∃``/``∧``/``∨``/``¬``/
21
+ ``→``/``⊕`` etc. all appear in the verified sample and all parse under this
22
+ kit's default ``fol`` mode).
23
+ * ``"conclusion"`` — ``str``, the natural-language conclusion.
24
+ * ``"conclusion-FOL"`` — ``str``, its gold FOL annotation.
25
+ * ``"label"`` — ``str``, the gold entailment label. The verified
26
+ sample showed ``"True"`` and ``"Uncertain"``; the paper and repository
27
+ README additionally document ``"False"`` as the third value (NOT directly
28
+ observed in the two rows fetched for this verification — flagged here
29
+ rather than silently assumed).
30
+
31
+ The verified ``v0.0`` rows carry NO id field at all (no ``example-id``,
32
+ ``story-id``, or ``source`` key was present in the fetched sample, contrary
33
+ to a fields description on the Hugging Face mirror's README, which this
34
+ adapter could not reach past the login gate to cross-check). This loader is
35
+ defensive rather than assuming either shape is final: it uses an
36
+ ``"example-id"``/``"example_id"`` key WHEN PRESENT, and otherwise falls back
37
+ to a positional id ``f"folio:{line_no}"`` (0-based line number within the
38
+ file) so every example is still addressable. Deliberately NOT used for id
39
+ resolution: ``"story-id"``/``"story_id"`` — the verified sample shows several
40
+ consecutive rows sharing identical ``premises``/``premises-FOL`` (one story,
41
+ several conclusions), so a story id identifies a GROUP of examples, not one
42
+ example, and using it alone as an id would collide across that group. It is
43
+ still preserved (like every other unrecognised key) in ``meta`` when present.
44
+
45
+ License: **CC-BY-SA-4.0**, per the ``LICENSE`` file at the repository root
46
+ (https://github.com/Yale-LILY/FOLIO/blob/main/LICENSE, verified directly).
47
+ Share-alike: adaptations/redistributions of the data itself must carry a
48
+ compatible license and attribution — this loader only reads a LOCAL file the
49
+ caller already obtained, it does not redistribute or embed any FOLIO data.
50
+
51
+ This module never downloads anything — obtain the JSONL file yourself from
52
+ one of the sources above and pass its local path to :func:`load_folio`.
53
+ """
54
+
55
+ import json
56
+ from pathlib import Path
57
+ from typing import FrozenSet, Iterator, Union
58
+
59
+ from ._base import DatasetExample, _register_dataset_info
60
+
61
+ __all__ = ["load_folio"]
62
+
63
+ _register_dataset_info(
64
+ "folio",
65
+ license="CC-BY-SA-4.0",
66
+ source_url="https://github.com/Yale-LILY/FOLIO",
67
+ citation_hint=(
68
+ "Han, Simin, et al. \"FOLIO: Natural Language Reasoning with "
69
+ "First-Order Logic.\" arXiv:2209.00840."
70
+ ),
71
+ )
72
+
73
+
74
+ def _resolve_id(record: dict, line_no: int) -> str:
75
+ """The record's own per-example id field if present, else a positional
76
+ fallback.
77
+
78
+ Tries both the hyphenated spelling documented in the field-description
79
+ text and the underscore spelling a JSON-key-safe re-export might use;
80
+ verified real data has neither (see module docstring), so the fallback
81
+ path is what actually fires against the canonical GitHub JSONL today.
82
+ Deliberately does NOT consult ``"story-id"``/``"story_id"`` — see the
83
+ module docstring for why that would not be a safe per-example id.
84
+ """
85
+ for key in ("example-id", "example_id"):
86
+ value = record.get(key)
87
+ if value is not None:
88
+ return str(value)
89
+ return f"folio:{line_no}"
90
+
91
+
92
+ def _example_from_record(record: dict, line_no: int,
93
+ known_bad_ids: FrozenSet[str]) -> DatasetExample:
94
+ premises = tuple(record.get("premises") or ())
95
+ premises_fol = tuple(record.get("premises-FOL") or ())
96
+ conclusion = record.get("conclusion")
97
+ conclusion_fol = record.get("conclusion-FOL")
98
+ label = record.get("label")
99
+ example_id = _resolve_id(record, line_no)
100
+
101
+ meta = {
102
+ k: v for k, v in record.items()
103
+ if k not in ("premises", "premises-FOL", "conclusion",
104
+ "conclusion-FOL", "label")
105
+ }
106
+ meta["line_no"] = line_no
107
+
108
+ return DatasetExample(
109
+ id=example_id,
110
+ nl_premises=premises,
111
+ fol_premises=premises_fol,
112
+ nl_conclusion=conclusion,
113
+ fol_conclusion=conclusion_fol,
114
+ label=label,
115
+ known_bad=example_id in known_bad_ids,
116
+ meta=meta,
117
+ )
118
+
119
+
120
+ def load_folio(path: Union[str, Path], *,
121
+ known_bad_ids: FrozenSet[str] = frozenset()) -> Iterator[DatasetExample]:
122
+ """Stream :class:`~unicode_logic_kit.eval.datasets.DatasetExample` from a
123
+ local FOLIO JSONL file.
124
+
125
+ Args:
126
+ path: path to a local ``.jsonl`` file in the verified FOLIO schema
127
+ (see module docstring) — one JSON object per non-blank line.
128
+ NEVER downloaded by this function; obtain the file from
129
+ https://github.com/Yale-LILY/FOLIO yourself.
130
+ known_bad_ids: ids (see :func:`_resolve_id`) whose gold annotation is
131
+ known to be broken (e.g. from a prior human review or an
132
+ :func:`~unicode_logic_kit.eval.datasets.audit_examples` pass on an
133
+ earlier load). Every yielded example with a matching id gets
134
+ ``known_bad=True``; everything else gets ``known_bad=False``.
135
+ Defaults to an empty set (nothing pre-flagged).
136
+
137
+ Yields:
138
+ One :class:`~unicode_logic_kit.eval.datasets.DatasetExample` per
139
+ non-blank JSONL line, in file order. ``premises``/``premises-FOL``
140
+ (mapped to ``nl_premises``/``fol_premises``) are guaranteed to be the
141
+ same length only insofar as the source file guarantees it — this
142
+ loader does not enforce or re-check that pairing (a length mismatch
143
+ would show up as a defect during evaluation, not silently here).
144
+
145
+ Raises:
146
+ FileNotFoundError: ``path`` does not exist.
147
+ json.JSONDecodeError: a non-blank line is not valid JSON — this is
148
+ NOT swallowed; a malformed dataset file is a loud failure, not a
149
+ silently-skipped row.
150
+ """
151
+ path = Path(path)
152
+ with path.open("r", encoding="utf-8") as fh:
153
+ for line_no, raw_line in enumerate(fh):
154
+ line = raw_line.strip()
155
+ if not line:
156
+ continue
157
+ record = json.loads(line)
158
+ yield _example_from_record(record, line_no, known_bad_ids)
@@ -0,0 +1,418 @@
1
+ """Adapter for the FraCaS textual-inference problem set — pure NLI, no FOL.
2
+
3
+ Every other adapter in this package carries gold FOL (or generates it from a
4
+ structured source). FraCaS carries NONE, by construction: its problems are
5
+ natural-language premises plus a hypothesis and a three-valued answer, and
6
+ there is no logic field anywhere in its DTD. That is exactly why it earns a
7
+ place here — the kit's entailment verdict is three-valued too, so FraCaS's
8
+ ``yes``/``no``/``unknown`` maps onto ``premises ⊨ h`` / ``premises ⊨ ¬h`` /
9
+ neither WITHOUT any interpretive glue, which makes it a reference target for
10
+ an NL→logic pipeline whose translation step lives OUTSIDE this library (see
11
+ :func:`solve_example`: the translation is an injected callable, never a
12
+ model this package calls).
13
+
14
+ Source and verified schema
15
+ ---------------------------
16
+ Verified 2026-08-19 directly against the canonical machine-readable edition,
17
+ ``https://nlp.stanford.edu/~wcmac/downloads/fracas.xml`` (XML conversion by
18
+ Bill MacCartney of the FraCaS Consortium's 1996 deliverable "Using the
19
+ Framework", Cooper et al.). Like every loader in this package it reads a
20
+ LOCAL file the caller already obtained — nothing here downloads anything.
21
+
22
+ The file's own header documents the representation; the numbers below are
23
+ this adapter's independent re-measurement of the file it parses:
24
+
25
+ * **346 problems**, ids ``"001"`` … ``"346"``, unique, zero-padded to three
26
+ digits (this adapter prefixes them: ``"fracas:001"``).
27
+ * **536 premises**, as ``<p idx="n">`` children — verified contiguous and
28
+ 1-based in every problem (192 problems have one premise, 122 two, 29
29
+ three, 2 four, 1 five). Read in ``idx`` order, not document order.
30
+ * ``<q>`` the original question, ``<h>`` the declarative hypothesis, ``<a>``
31
+ the source document's answer text (``"Yes"``, ``"Don't know"``, but also
32
+ qualified phrases like ``"Not many"``), optional ``<why>`` (110) and
33
+ ``<note>`` (33).
34
+ * ``fracas_answer`` ∈ ``yes`` (203) / ``unknown`` (98) / ``no`` (33) /
35
+ ``undef`` (12) — the canonicalisation of ``<a>``; this is the ``label``.
36
+ * ``fracas_nonstandard="true"`` on the 41 problems whose ``<a>`` is not one
37
+ of the three canonical answers.
38
+ * Sections are NOT attributes: they are ``<comment class="section">`` /
39
+ ``"subsection"`` / ``"subsubsection"`` markers between problems (9 / 47 /
40
+ 10 of them), so section membership is DOCUMENT ORDER. This adapter tracks
41
+ them as it walks and resets the finer levels whenever a coarser one
42
+ changes — a problem can therefore never inherit a stale subsection from
43
+ the previous section.
44
+
45
+ Honest limitations
46
+ -------------------
47
+ * **Four problems (276, 305, 309, 310) have an EMPTY ``<q>`` and ``<h>``** —
48
+ the source document has no question for them. They load (nothing is
49
+ dropped silently) with ``nl_conclusion=None`` and are refused by
50
+ :func:`solve_example` with a named error rather than scored against an
51
+ absent hypothesis. All four are also ``undef``.
52
+ * **``undef`` is not a fourth answer class**, it marks a problem whose
53
+ source answer is not canonicalisable at all. Such examples load with
54
+ ``label="undef"``; :func:`solve_example` will still PREDICT for the eight
55
+ of them that have a hypothesis (predicting is not scoring), and the caller
56
+ is expected to exclude them from any accuracy figure.
57
+ * ``fol_premises`` is always empty and ``fol_conclusion`` always ``None``:
58
+ there is no gold FOL to audit, so
59
+ :func:`~unicode_logic_kit.eval.datasets.audit_examples` reports these
60
+ examples as ``ok`` VACUOUSLY. That is not a claim about the data.
61
+ * The answer text in ``<a>`` is kept verbatim in ``meta["answer_text"]``,
62
+ including the qualified ones — canonicalising them further would be this
63
+ adapter inventing gold labels.
64
+ """
65
+
66
+ import re
67
+ import xml.etree.ElementTree as ET
68
+ from pathlib import Path
69
+ from typing import (
70
+ Callable, Dict, FrozenSet, Iterable, Iterator, List, Optional, Union,
71
+ )
72
+
73
+ from ._base import DatasetExample, _register_dataset_info
74
+
75
+ __all__ = ["load_fracas", "solve_example", "ace_census", "FRACAS_ANSWERS"]
76
+
77
+
78
+ #: The canonical values of the ``fracas_answer`` attribute. ``undef`` is a
79
+ #: "no canonical answer exists" marker, not a fourth answer — see the module
80
+ #: docstring.
81
+ FRACAS_ANSWERS = ("yes", "no", "unknown", "undef")
82
+
83
+ _SECTION_LEVELS = ("section", "subsection", "subsubsection")
84
+ _HEADING_RE = re.compile(r"^([\d.]+)\s+(.*)$", re.DOTALL)
85
+
86
+
87
+ _register_dataset_info(
88
+ "fracas",
89
+ license=("no explicit licence statement in the source file; the XML "
90
+ "edition asks for credit for the conversion, and the problems "
91
+ "derive from the FraCaS Consortium's 1996 deliverable"),
92
+ source_url="https://nlp.stanford.edu/~wcmac/downloads/fracas.xml",
93
+ citation_hint=('FraCaS Consortium (Cooper et al.), "Using the '
94
+ 'Framework", 1996; XML edition by Bill MacCartney.'),
95
+ )
96
+
97
+
98
+ # ---------------------------------------------------------------------------
99
+ # Reading
100
+ # ---------------------------------------------------------------------------
101
+
102
+ def _text(element: Optional[ET.Element]) -> Optional[str]:
103
+ """Element text with XML indentation collapsed — ``None`` when absent or
104
+ empty (the four question-less problems), never the empty string."""
105
+ if element is None or element.text is None:
106
+ return None
107
+ collapsed = " ".join(element.text.split())
108
+ return collapsed or None
109
+
110
+
111
+ def _heading(raw: Optional[str]) -> Dict[str, Optional[str]]:
112
+ """``"1.2 Monotonicity (…)"`` → number and title, kept separate so a
113
+ caller filters on the STABLE number rather than on prose."""
114
+ if raw is None:
115
+ return {"number": None, "title": None}
116
+ match = _HEADING_RE.match(raw)
117
+ if match is None:
118
+ return {"number": None, "title": raw}
119
+ return {"number": match.group(1), "title": " ".join(match.group(2).split())}
120
+
121
+
122
+ def _premises(problem: ET.Element) -> List[str]:
123
+ """The ``<p>`` texts in ``idx`` order, with the ordering CHECKED: the
124
+ file's indices are contiguous and 1-based throughout, so anything else
125
+ is a corrupted input and says so instead of being silently reordered."""
126
+ numbered = []
127
+ for element in problem.findall("p"):
128
+ raw_idx = element.get("idx")
129
+ if raw_idx is None or not raw_idx.isdigit():
130
+ raise ValueError(
131
+ f"fracas: problem {problem.get('id')!r} has a <p> without a "
132
+ f"numeric idx (got {raw_idx!r})")
133
+ text = _text(element)
134
+ if text is None:
135
+ raise ValueError(
136
+ f"fracas: problem {problem.get('id')!r} has an empty premise "
137
+ f"at idx {raw_idx}")
138
+ numbered.append((int(raw_idx), text))
139
+ numbered.sort()
140
+ if [i for i, _ in numbered] != list(range(1, len(numbered) + 1)):
141
+ raise ValueError(
142
+ f"fracas: problem {problem.get('id')!r} has non-contiguous "
143
+ f"premise indices {[i for i, _ in numbered]}")
144
+ return [text for _, text in numbered]
145
+
146
+
147
+ def load_fracas(path: Union[str, Path], *,
148
+ sections: Optional[Iterable[str]] = None,
149
+ answers: Optional[Iterable[str]] = None,
150
+ known_bad_ids: FrozenSet[str] = frozenset(),
151
+ ) -> Iterator[DatasetExample]:
152
+ """Read the FraCaS XML into :class:`DatasetExample` objects, in file order.
153
+
154
+ Field mapping (see the module docstring for what each source element is):
155
+ ``nl_premises`` = the ``<p>`` texts in ``idx`` order, ``nl_conclusion`` =
156
+ ``<h>`` (``None`` for the four question-less problems), ``label`` =
157
+ ``fracas_answer``, and ``fol_premises``/``fol_conclusion`` stay empty —
158
+ FraCaS has no logic annotation. Everything else from the record survives
159
+ in ``meta``: ``question``, ``answer_text``, ``why``, ``note``,
160
+ ``nonstandard``, ``premise_count``, and the section / subsection /
161
+ subsubsection numbers and titles.
162
+
163
+ Args:
164
+ path: the local ``fracas.xml``.
165
+ sections: keep only problems in these SECTION NUMBERS (``{"1", "3"}``
166
+ — the stable identifier, matched against the top-level section,
167
+ so ``"1"`` keeps all of ``1.x``). ``None`` keeps everything.
168
+ answers: keep only these ``fracas_answer`` values (e.g.
169
+ ``{"yes", "no", "unknown"}`` to drop the twelve ``undef``
170
+ problems). ``None`` keeps everything, ``undef`` included — this
171
+ loader never drops them on its own.
172
+ known_bad_ids: ids (in the prefixed ``"fracas:001"`` form) to flag as
173
+ ``known_bad``; the same caller-curated mechanic every adapter has.
174
+
175
+ Raises:
176
+ ValueError: the file is not a FraCaS problem set, a problem lacks its
177
+ id or ``fracas_answer``, an answer is outside
178
+ :data:`FRACAS_ANSWERS`, ids repeat, or premise indices are not
179
+ contiguous — a malformed input is named, never worked around.
180
+ """
181
+ wanted_sections = None if sections is None else {str(s) for s in sections}
182
+ wanted_answers = None if answers is None else {str(a) for a in answers}
183
+ if wanted_answers is not None:
184
+ unknown = wanted_answers - set(FRACAS_ANSWERS)
185
+ if unknown:
186
+ raise ValueError(
187
+ f"fracas: answers={sorted(unknown)} is outside "
188
+ f"{list(FRACAS_ANSWERS)}")
189
+
190
+ root = ET.parse(str(path)).getroot()
191
+ if root.tag != "fracas-problems":
192
+ raise ValueError(
193
+ f"fracas: {path} has root element {root.tag!r}, expected "
194
+ "'fracas-problems' — is this the FraCaS XML?")
195
+
196
+ headings: Dict[str, Dict[str, Optional[str]]] = {
197
+ level: _heading(None) for level in _SECTION_LEVELS}
198
+ seen = set()
199
+
200
+ for element in root:
201
+ if element.tag == "comment":
202
+ level = element.get("class")
203
+ if level in _SECTION_LEVELS:
204
+ headings[level] = _heading(_text(element))
205
+ # A coarser heading invalidates every finer one, so a
206
+ # problem can never inherit a stale subsection.
207
+ for finer in _SECTION_LEVELS[_SECTION_LEVELS.index(level) + 1:]:
208
+ headings[finer] = _heading(None)
209
+ continue
210
+ if element.tag != "problem":
211
+ continue
212
+
213
+ raw_id = element.get("id")
214
+ if not raw_id:
215
+ raise ValueError("fracas: a <problem> element has no id")
216
+ example_id = f"fracas:{raw_id}"
217
+ if example_id in seen:
218
+ raise ValueError(f"fracas: duplicate problem id {raw_id!r}")
219
+ seen.add(example_id)
220
+
221
+ answer = element.get("fracas_answer")
222
+ if answer is None:
223
+ raise ValueError(
224
+ f"fracas: problem {raw_id!r} has no fracas_answer attribute")
225
+ if answer not in FRACAS_ANSWERS:
226
+ raise ValueError(
227
+ f"fracas: problem {raw_id!r} has fracas_answer {answer!r}, "
228
+ f"outside {list(FRACAS_ANSWERS)}")
229
+
230
+ if (wanted_sections is not None
231
+ and headings["section"]["number"] not in wanted_sections):
232
+ continue
233
+ if wanted_answers is not None and answer not in wanted_answers:
234
+ continue
235
+
236
+ premises = _premises(element)
237
+ meta = {
238
+ "question": _text(element.find("q")),
239
+ "answer_text": _text(element.find("a")),
240
+ "why": _text(element.find("why")),
241
+ "note": _text(element.find("note")),
242
+ "nonstandard": element.get("fracas_nonstandard") == "true",
243
+ "premise_count": len(premises),
244
+ }
245
+ for level in _SECTION_LEVELS:
246
+ meta[level] = headings[level]["number"]
247
+ meta[f"{level}_title"] = headings[level]["title"]
248
+
249
+ yield DatasetExample(
250
+ id=example_id,
251
+ nl_premises=tuple(premises),
252
+ fol_premises=(),
253
+ nl_conclusion=_text(element.find("h")),
254
+ fol_conclusion=None,
255
+ label=answer,
256
+ known_bad=example_id in known_bad_ids,
257
+ meta=meta,
258
+ )
259
+
260
+
261
+ # ---------------------------------------------------------------------------
262
+ # Deciding — with the translation injected by the caller
263
+ # ---------------------------------------------------------------------------
264
+
265
+ def solve_example(example: DatasetExample, *,
266
+ translate: Callable[[str], object],
267
+ on_indefinite: str = "label", **prove_kwargs) -> dict:
268
+ """Decide one FraCaS problem end-to-end — the translation is YOURS.
269
+
270
+ FraCaS ships no formulas, so this helper takes ``translate``: a callable
271
+ mapping one natural-language sentence to either a formula string (parsed
272
+ with :func:`unicode_logic_kit.api.parse_any`) or an already-built kit node.
273
+ That is the seam where an external system — a semantic parser, a
274
+ hand-written table, a language model driven by the caller — plugs in;
275
+ this package deliberately calls no such system itself.
276
+
277
+ The rest is the same three-valued cascade the other adapters use, and it
278
+ matches FraCaS's own answer semantics exactly: ``"yes"`` iff premises ⊨
279
+ hypothesis, ``"no"`` iff premises ⊨ ¬hypothesis, ``"unknown"`` otherwise.
280
+ Extra ``prove_kwargs`` reach :func:`unicode_logic_kit.api.prove` verbatim,
281
+ so the prover is the caller's choice.
282
+
283
+ ``on_indefinite`` handles a NON-DEFINITIVE prover outcome (timeout, hit
284
+ bound, honest incompleteness) when neither direction was proved:
285
+
286
+ - ``"label"`` (default): predict ``"unknown"``.
287
+ - ``"abstain"``: ``"unknown"`` only when BOTH directions came back
288
+ definitively refuted (underdetermination established by countermodels);
289
+ any indefinite leg yields ``predicted=None``, so a timeout can never be
290
+ scored as a correct "unknown".
291
+ - ``"raise"``: like ``"abstain"`` but raises instead.
292
+
293
+ Returns a dict with ``predicted`` (``"yes"``/``"no"``/``"unknown"``/
294
+ ``None``), ``label`` (the gold answer, ``"undef"`` included — scoring
295
+ against it is the caller's decision), ``verdict``/``verdict_negated``
296
+ (verdict dicts; the negated one is ``None`` when the positive direction
297
+ already settled it), and the translated ``premises``/``hypothesis`` in
298
+ kit notation, so a wrong prediction can be traced back to the
299
+ translation that caused it.
300
+
301
+ Raises:
302
+ ValueError: the example has no hypothesis (the four question-less
303
+ problems), ``on_indefinite`` is not one of the three modes, or a
304
+ translated string does not parse.
305
+ """
306
+ from ... import api
307
+ from ...fol.nodes import Node, Not
308
+
309
+ if on_indefinite not in ("label", "abstain", "raise"):
310
+ raise ValueError(
311
+ f"fracas: on_indefinite must be 'label', 'abstain' or 'raise', "
312
+ f"got {on_indefinite!r}")
313
+ if example.nl_conclusion is None:
314
+ raise ValueError(
315
+ f"fracas: example {example.id} has no hypothesis (the source "
316
+ "document has no question for it) — nothing to decide.")
317
+
318
+ def _formula(sentence: str) -> "Node":
319
+ produced = translate(sentence)
320
+ if isinstance(produced, Node):
321
+ return produced
322
+ if not isinstance(produced, str):
323
+ raise ValueError(
324
+ f"fracas: example {example.id}: translate({sentence!r}) "
325
+ f"returned {type(produced).__name__}, expected a formula "
326
+ "string or a kit node")
327
+ parsed = api.parse_any(produced)
328
+ if not parsed.ok:
329
+ raise ValueError(
330
+ f"fracas: example {example.id}: the translation "
331
+ f"{produced!r} of {sentence!r} does not parse")
332
+ return parsed.formula
333
+
334
+ premises = [_formula(sentence) for sentence in example.nl_premises]
335
+ hypothesis = _formula(example.nl_conclusion)
336
+
337
+ result = {
338
+ "label": example.label,
339
+ "premises": [p.to_unicode_str() for p in premises],
340
+ "hypothesis": hypothesis.to_unicode_str(),
341
+ }
342
+ verdict = api.prove(hypothesis, premises, **prove_kwargs)
343
+ if verdict.status == "proved":
344
+ result.update(predicted="yes", verdict=verdict.to_dict(),
345
+ verdict_negated=None)
346
+ return result
347
+
348
+ negated = api.prove(Not(hypothesis), premises, **prove_kwargs)
349
+ if negated.status == "proved":
350
+ predicted: Optional[str] = "no"
351
+ elif on_indefinite == "label":
352
+ predicted = "unknown"
353
+ elif verdict.status == "refuted" and negated.status == "refuted":
354
+ # Underdetermination ESTABLISHED both ways: "unknown" is a definitive
355
+ # answer here, so even abstain/raise report it.
356
+ predicted = "unknown"
357
+ elif on_indefinite == "raise":
358
+ raise ValueError(
359
+ f"fracas: example {example.id}: indefinite prover outcome "
360
+ f"(goal: {verdict.status}/{verdict.reason}, negated: "
361
+ f"{negated.status}/{negated.reason}) with on_indefinite='raise'.")
362
+ else: # "abstain"
363
+ predicted = None
364
+
365
+ result.update(predicted=predicted, verdict=verdict.to_dict(),
366
+ verdict_negated=negated.to_dict())
367
+ return result
368
+
369
+
370
+ # ---------------------------------------------------------------------------
371
+ # How much of it is controlled English?
372
+ # ---------------------------------------------------------------------------
373
+
374
+ def ace_census(examples: Iterable[DatasetExample], *,
375
+ ulex: Optional[str] = None, timeout: float = 30.0,
376
+ ) -> List[dict]:
377
+ """Per-SENTENCE report: which FraCaS sentences does APE accept as ACE?
378
+
379
+ A measurement, not a score. FraCaS is short, deliberately plain English,
380
+ so it is the natural corpus for asking how far Attempto Controlled
381
+ English reaches as a target notation — and the answer comes per
382
+ sentence, with APE's own diagnosis attached, never as a single aggregate
383
+ this function decides for you (group the rows by ``section`` yourself).
384
+
385
+ Needs a reachable APE binary
386
+ (:func:`unicode_logic_kit.ace.ape_available`); ``ulex`` is passed through
387
+ as APE's user lexicon, which matters because APE's built-in lexicon is
388
+ small and a missing word is reported as "not ACE" like any other
389
+ refusal.
390
+
391
+ Returns one dict per sentence, in example order: ``id``, ``section``,
392
+ ``role`` (``"premise"``/``"hypothesis"``), ``index`` (position within the
393
+ premises, ``None`` for the hypothesis), ``sentence``, ``status``
394
+ (:class:`~unicode_logic_kit.ace.runner.CoverageRow`'s vocabulary:
395
+ ``ok``/``tptp_unsupported``/``tptp_unread``/``not_ace``/``infra``) and
396
+ ``detail``.
397
+ """
398
+ from ...ace import ace_coverage
399
+
400
+ rows: List[dict] = []
401
+ for example in examples:
402
+ sentences = [("premise", i, s)
403
+ for i, s in enumerate(example.nl_premises)]
404
+ if example.nl_conclusion is not None:
405
+ sentences.append(("hypothesis", None, example.nl_conclusion))
406
+ coverage = ace_coverage([s for _, _, s in sentences],
407
+ ulex=ulex, timeout=timeout)
408
+ for (role, index, sentence), row in zip(sentences, coverage):
409
+ rows.append({
410
+ "id": example.id,
411
+ "section": example.meta.get("section"),
412
+ "role": role,
413
+ "index": index,
414
+ "sentence": sentence,
415
+ "status": row.status,
416
+ "detail": row.detail,
417
+ })
418
+ return rows