unicode-logic-kit 0.31.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. unicode_logic_kit/__init__.py +385 -0
  2. unicode_logic_kit/__main__.py +520 -0
  3. unicode_logic_kit/_deadline.py +219 -0
  4. unicode_logic_kit/ace/__init__.py +126 -0
  5. unicode_logic_kit/ace/_align.py +135 -0
  6. unicode_logic_kit/ace/chem_lexicon.py +128 -0
  7. unicode_logic_kit/ace/drs_reader.py +570 -0
  8. unicode_logic_kit/ace/mapping.py +666 -0
  9. unicode_logic_kit/ace/reverse_modal.py +138 -0
  10. unicode_logic_kit/ace/runner.py +551 -0
  11. unicode_logic_kit/ace/translate.py +452 -0
  12. unicode_logic_kit/ace/verbalize.py +1070 -0
  13. unicode_logic_kit/api.py +1284 -0
  14. unicode_logic_kit/atp/__init__.py +177 -0
  15. unicode_logic_kit/atp/_ascii_names.py +113 -0
  16. unicode_logic_kit/atp/_html.py +72 -0
  17. unicode_logic_kit/atp/_substructural_input.py +228 -0
  18. unicode_logic_kit/atp/_tff_problem.py +715 -0
  19. unicode_logic_kit/atp/_tptp_problem.py +1111 -0
  20. unicode_logic_kit/atp/_writer_support.py +289 -0
  21. unicode_logic_kit/atp/clingo_backend.py +1180 -0
  22. unicode_logic_kit/atp/cvc5_backend.py +1385 -0
  23. unicode_logic_kit/atp/eprover_backend.py +732 -0
  24. unicode_logic_kit/atp/finite_domain.py +1055 -0
  25. unicode_logic_kit/atp/fitch.py +1547 -0
  26. unicode_logic_kit/atp/fitch_search.py +551 -0
  27. unicode_logic_kit/atp/hets_backend.py +339 -0
  28. unicode_logic_kit/atp/hybrid_down.py +120 -0
  29. unicode_logic_kit/atp/incremental.py +250 -0
  30. unicode_logic_kit/atp/kripke_enum.py +741 -0
  31. unicode_logic_kit/atp/lambek.py +436 -0
  32. unicode_logic_kit/atp/leo3_backend.py +332 -0
  33. unicode_logic_kit/atp/linear.py +738 -0
  34. unicode_logic_kit/atp/lj.py +705 -0
  35. unicode_logic_kit/atp/logic_backends.py +566 -0
  36. unicode_logic_kit/atp/ltl_tableau.py +1084 -0
  37. unicode_logic_kit/atp/minizinc_backend.py +1402 -0
  38. unicode_logic_kit/atp/modal_tableau.py +1382 -0
  39. unicode_logic_kit/atp/nanocop_backend.py +410 -0
  40. unicode_logic_kit/atp/portfolio.py +489 -0
  41. unicode_logic_kit/atp/protocol.py +1803 -0
  42. unicode_logic_kit/atp/prover9_entailment.py +1153 -0
  43. unicode_logic_kit/atp/resolution.py +1376 -0
  44. unicode_logic_kit/atp/resolution_check.py +1114 -0
  45. unicode_logic_kit/atp/sequent.py +1050 -0
  46. unicode_logic_kit/atp/tableau.py +921 -0
  47. unicode_logic_kit/atp/tableau_check.py +543 -0
  48. unicode_logic_kit/atp/tptp_ncl.py +811 -0
  49. unicode_logic_kit/atp/tptp_tff.py +1546 -0
  50. unicode_logic_kit/atp/tstp.py +1333 -0
  51. unicode_logic_kit/atp/tstp_check.py +1096 -0
  52. unicode_logic_kit/atp/twee_backend.py +236 -0
  53. unicode_logic_kit/atp/twee_check.py +711 -0
  54. unicode_logic_kit/atp/twee_entailment.py +953 -0
  55. unicode_logic_kit/atp/vampire_entailment.py +540 -0
  56. unicode_logic_kit/atp/z3_arith.py +470 -0
  57. unicode_logic_kit/atp/z3_equivalence.py +36 -0
  58. unicode_logic_kit/atp/z3_fuzzy.py +362 -0
  59. unicode_logic_kit/atp/z3_input.py +500 -0
  60. unicode_logic_kit/atp/z3_models.py +208 -0
  61. unicode_logic_kit/chem/__init__.py +88 -0
  62. unicode_logic_kit/chem/_naming.py +284 -0
  63. unicode_logic_kit/chem/cache.py +185 -0
  64. unicode_logic_kit/chem/interop.py +244 -0
  65. unicode_logic_kit/chem/mol.py +525 -0
  66. unicode_logic_kit/chem/signature.py +112 -0
  67. unicode_logic_kit/comorphism.py +497 -0
  68. unicode_logic_kit/dl/__init__.py +384 -0
  69. unicode_logic_kit/dl/classification.py +227 -0
  70. unicode_logic_kit/dl/concepts.py +632 -0
  71. unicode_logic_kit/dl/datatypes.py +818 -0
  72. unicode_logic_kit/dl/owl_functional.py +2433 -0
  73. unicode_logic_kit/dl/owl_manchester.py +1637 -0
  74. unicode_logic_kit/dl/owl_reasoner.py +790 -0
  75. unicode_logic_kit/dl/parser.py +391 -0
  76. unicode_logic_kit/dl/tableau.py +4048 -0
  77. unicode_logic_kit/dl/translate.py +2704 -0
  78. unicode_logic_kit/drt/__init__.py +94 -0
  79. unicode_logic_kit/drt/export.py +179 -0
  80. unicode_logic_kit/drt/nodes.py +506 -0
  81. unicode_logic_kit/drt/parser.py +965 -0
  82. unicode_logic_kit/drt/resolve.py +195 -0
  83. unicode_logic_kit/drt/reverse.py +175 -0
  84. unicode_logic_kit/eval/__init__.py +106 -0
  85. unicode_logic_kit/eval/batch.py +382 -0
  86. unicode_logic_kit/eval/canonical.py +663 -0
  87. unicode_logic_kit/eval/chem_batch.py +606 -0
  88. unicode_logic_kit/eval/converses.py +200 -0
  89. unicode_logic_kit/eval/datasets/__init__.py +136 -0
  90. unicode_logic_kit/eval/datasets/_base.py +263 -0
  91. unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
  92. unicode_logic_kit/eval/datasets/c3po.py +678 -0
  93. unicode_logic_kit/eval/datasets/folio.py +158 -0
  94. unicode_logic_kit/eval/datasets/fracas.py +418 -0
  95. unicode_logic_kit/eval/datasets/groves.py +191 -0
  96. unicode_logic_kit/eval/datasets/logicbench.py +467 -0
  97. unicode_logic_kit/eval/datasets/logicnli.py +303 -0
  98. unicode_logic_kit/eval/datasets/malls.py +133 -0
  99. unicode_logic_kit/eval/datasets/pfolio.py +594 -0
  100. unicode_logic_kit/eval/datasets/pmb.py +242 -0
  101. unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
  102. unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
  103. unicode_logic_kit/eval/datasets/proverqa.py +674 -0
  104. unicode_logic_kit/eval/datasets/willow.py +478 -0
  105. unicode_logic_kit/eval/equivalence.py +466 -0
  106. unicode_logic_kit/eval/exercise_gen.py +533 -0
  107. unicode_logic_kit/eval/explain.py +791 -0
  108. unicode_logic_kit/eval/generality.py +750 -0
  109. unicode_logic_kit/eval/metric_hf.py +458 -0
  110. unicode_logic_kit/eval/predicate_match.py +343 -0
  111. unicode_logic_kit/eval/theory_check.py +1170 -0
  112. unicode_logic_kit/eval/validate.py +306 -0
  113. unicode_logic_kit/fol/__init__.py +177 -0
  114. unicode_logic_kit/fol/_atom_keys.py +510 -0
  115. unicode_logic_kit/fol/_fol_nodes.py +3586 -0
  116. unicode_logic_kit/fol/_free_parameters.py +105 -0
  117. unicode_logic_kit/fol/_ho_nodes.py +448 -0
  118. unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
  119. unicode_logic_kit/fol/_identifiers.py +1091 -0
  120. unicode_logic_kit/fol/_lambek_nodes.py +112 -0
  121. unicode_logic_kit/fol/_linear_nodes.py +352 -0
  122. unicode_logic_kit/fol/_modal_nodes.py +1467 -0
  123. unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
  124. unicode_logic_kit/fol/_numeral_symbols.py +231 -0
  125. unicode_logic_kit/fol/_so_nodes.py +200 -0
  126. unicode_logic_kit/fol/_symbol_names.py +81 -0
  127. unicode_logic_kit/fol/_team_nodes.py +181 -0
  128. unicode_logic_kit/fol/_tptp_symbols.py +551 -0
  129. unicode_logic_kit/fol/_truth_constants.py +117 -0
  130. unicode_logic_kit/fol/casl_export.py +1135 -0
  131. unicode_logic_kit/fol/casl_import.py +929 -0
  132. unicode_logic_kit/fol/derivation.py +367 -0
  133. unicode_logic_kit/fol/dialect_detect.py +70 -0
  134. unicode_logic_kit/fol/dialect_repair.py +537 -0
  135. unicode_logic_kit/fol/frames.py +637 -0
  136. unicode_logic_kit/fol/grammars/terminals.lark +31 -0
  137. unicode_logic_kit/fol/lambda_tools.py +297 -0
  138. unicode_logic_kit/fol/latex_input.py +429 -0
  139. unicode_logic_kit/fol/modal_translation.py +944 -0
  140. unicode_logic_kit/fol/msflparser.py +1033 -0
  141. unicode_logic_kit/fol/naming.py +422 -0
  142. unicode_logic_kit/fol/nodes.py +241 -0
  143. unicode_logic_kit/fol/normalforms.py +492 -0
  144. unicode_logic_kit/fol/pal.py +287 -0
  145. unicode_logic_kit/fol/prolog_export.py +566 -0
  146. unicode_logic_kit/fol/prolog_input.py +505 -0
  147. unicode_logic_kit/fol/prover9_input.py +1325 -0
  148. unicode_logic_kit/fol/qml.py +1760 -0
  149. unicode_logic_kit/fol/qmltp_input.py +525 -0
  150. unicode_logic_kit/fol/sanitize.py +221 -0
  151. unicode_logic_kit/fol/serialize.py +79 -0
  152. unicode_logic_kit/fol/signature.py +1290 -0
  153. unicode_logic_kit/fol/simplify_check.py +544 -0
  154. unicode_logic_kit/fol/spans.py +594 -0
  155. unicode_logic_kit/fol/tptp_input.py +1503 -0
  156. unicode_logic_kit/fol/tptp_repair.py +941 -0
  157. unicode_logic_kit/fol/unification.py +157 -0
  158. unicode_logic_kit/fol/verbalize.py +263 -0
  159. unicode_logic_kit/hets/__init__.py +163 -0
  160. unicode_logic_kit/hets/bridge.py +142 -0
  161. unicode_logic_kit/hets/client.py +748 -0
  162. unicode_logic_kit/hets/docker.py +420 -0
  163. unicode_logic_kit/hets/dol.py +712 -0
  164. unicode_logic_kit/hets/haskell_json.py +355 -0
  165. unicode_logic_kit/hets/owl_backend.py +794 -0
  166. unicode_logic_kit/hets/owl_cli.py +598 -0
  167. unicode_logic_kit/hets/symbols.py +512 -0
  168. unicode_logic_kit/hol/__init__.py +140 -0
  169. unicode_logic_kit/hol/_ho_common.py +323 -0
  170. unicode_logic_kit/hol/_isabelle_binders.py +125 -0
  171. unicode_logic_kit/hol/classical.py +812 -0
  172. unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
  173. unicode_logic_kit/hol/deepshallow/_common.py +177 -0
  174. unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
  175. unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
  176. unicode_logic_kit/hol/deepshallow/modal.py +217 -0
  177. unicode_logic_kit/hol/deepshallow/qml.py +406 -0
  178. unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
  179. unicode_logic_kit/hol/free.py +753 -0
  180. unicode_logic_kit/hol/goedel.py +336 -0
  181. unicode_logic_kit/hol/ho_modal.py +1743 -0
  182. unicode_logic_kit/hol/intuitionistic.py +403 -0
  183. unicode_logic_kit/hol/isabelle_conditional.py +593 -0
  184. unicode_logic_kit/hol/isabelle_modal.py +1908 -0
  185. unicode_logic_kit/hol/isabelle_relevant.py +412 -0
  186. unicode_logic_kit/hol/isabelle_runner.py +1147 -0
  187. unicode_logic_kit/hol/isabelle_substructural.py +884 -0
  188. unicode_logic_kit/hol/lean.py +1018 -0
  189. unicode_logic_kit/hol/manyvalued.py +921 -0
  190. unicode_logic_kit/hol/secondorder.py +687 -0
  191. unicode_logic_kit/hol/thf_modal.py +941 -0
  192. unicode_logic_kit/hol/thirdorder.py +397 -0
  193. unicode_logic_kit/ilp/__init__.py +89 -0
  194. unicode_logic_kit/ilp/readback.py +389 -0
  195. unicode_logic_kit/ilp/separation.py +153 -0
  196. unicode_logic_kit/ilp/task.py +730 -0
  197. unicode_logic_kit/logic.py +163 -0
  198. unicode_logic_kit/mcp/__init__.py +28 -0
  199. unicode_logic_kit/mcp/__main__.py +5 -0
  200. unicode_logic_kit/mcp/chem_tools.py +1031 -0
  201. unicode_logic_kit/mcp/server.py +2453 -0
  202. unicode_logic_kit/mcp/syntax_spec.py +681 -0
  203. unicode_logic_kit/prob/__init__.py +53 -0
  204. unicode_logic_kit/prob/_bdd.py +225 -0
  205. unicode_logic_kit/prob/_column_gen.py +668 -0
  206. unicode_logic_kit/prob/distribution.py +686 -0
  207. unicode_logic_kit/prob/nilsson.py +470 -0
  208. unicode_logic_kit/py.typed +0 -0
  209. unicode_logic_kit/semantics/__init__.py +137 -0
  210. unicode_logic_kit/semantics/_modal_reject.py +156 -0
  211. unicode_logic_kit/semantics/action_models.py +466 -0
  212. unicode_logic_kit/semantics/asp_models.py +1200 -0
  213. unicode_logic_kit/semantics/conditional.py +580 -0
  214. unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
  215. unicode_logic_kit/semantics/free_logic.py +913 -0
  216. unicode_logic_kit/semantics/fuzzy.py +384 -0
  217. unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
  218. unicode_logic_kit/semantics/intuitionistic.py +581 -0
  219. unicode_logic_kit/semantics/kripke.py +1139 -0
  220. unicode_logic_kit/semantics/manyvalued.py +580 -0
  221. unicode_logic_kit/semantics/matrix.py +342 -0
  222. unicode_logic_kit/semantics/model_eval.py +1135 -0
  223. unicode_logic_kit/semantics/modelfinder.py +1036 -0
  224. unicode_logic_kit/semantics/nonmonotonic.py +372 -0
  225. unicode_logic_kit/semantics/relevant.py +331 -0
  226. unicode_logic_kit/semantics/secondorder.py +657 -0
  227. unicode_logic_kit/semantics/structures.py +352 -0
  228. unicode_logic_kit/semantics/tarski.py +975 -0
  229. unicode_logic_kit/semantics/team.py +315 -0
  230. unicode_logic_kit/semantics/team_translation.py +416 -0
  231. unicode_logic_kit/semantics/thirdorder.py +358 -0
  232. unicode_logic_kit/semantics/tnorm.py +85 -0
  233. unicode_logic_kit/semantics/truthtable.py +201 -0
  234. unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
  235. unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
  236. unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
  237. unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,429 @@
1
+ """LaTeX input: read a LaTeX math-mode formula and parse it with MSFLParser.
2
+
3
+ This module is the exact inverse of ``node.to_latex()`` (the LaTeX renderer in
4
+ ``_msfl_nodes.py``). It translates LaTeX math-mode markup into the toolkit's
5
+ Unicode surface syntax and then hands the result to :class:`MSFLParser`.
6
+
7
+ The translation is a tokenizing replacement, run as a fixed pipeline (see
8
+ :func:`latex_to_unicode` for the exact, numbered steps): multi-token brace
9
+ constructs (``\\mathbin{\\mathsf{U}}``, ``\\mathsf{G}``, ``\\mathrm{...}``,
10
+ ``{:}``, ``\\left(`` / ``\\right)``, subscript braces ``X_{...}``) are
11
+ resolved first; escaped literal braces (``\\{`` ``\\}``, used by the
12
+ cardinality and slashed-existential constructs) are protected from the later
13
+ brace-stripping step; a bare ``\\mathsf{name}`` left over after the specific
14
+ multi-token constructs is unwrapped as a hybrid-logic nominal; then backslash
15
+ control sequences (``\\leftrightarrow``, ``\\forall``, …) are mapped
16
+ glyph-for-glyph by matching the FULL ``[a-zA-Z]+`` run after the backslash (so
17
+ ``\\leq`` is never shadowed by ``\\le``); then LaTeX spacing (``\\,`` ``\\;``
18
+ ``\\!`` ``\\quad`` ``\\qquad`` and a backslash-space) is deleted; the counting
19
+ quantifier's exponent (``\\exists^{\\geq 3}``) is collapsed to the glued
20
+ ``∃≥3`` terminal; finally leftover grouping braces are stripped (the protected
21
+ literal ones are restored right after), since operator precedence in the
22
+ Unicode surface syntax is explicit and LaTeX grouping carries no information
23
+ the parser needs.
24
+
25
+ The control-sequence map also accepts the common hand-written synonyms a person
26
+ would type by hand (``\\neg`` for ``¬``, ``\\to`` for ``→``, ``\\iff`` for
27
+ ``↔``, ``\\le`` for ``≤``, ``\\times`` for ``*``, …) so that pasted LaTeX need
28
+ not have come from ``to_latex``.
29
+
30
+ A single quote is the one character this reader refuses. The Unicode syntax
31
+ writes a constant whose name is not a bare word in single quotes (``'k2'``,
32
+ ``'John Doe'``), and the name between the quotes is the name exactly; the
33
+ substitutions above know nothing of quotes, so they would collapse the spaces
34
+ of ``'John Doe'``, turn ``'a\\_b'`` into ``'a_b'`` and strip the braces of
35
+ ``'f{x}'`` without a word, and the formula would come back with another
36
+ constant. Until this reader tracks quotes it reads no text that holds one: a
37
+ :class:`LatexParsingError` names the quote and says to write the formula in the
38
+ Unicode syntax. A prime (``x'``) is such a quote. Before quoted constants it
39
+ was a syntax error of the parser (an unexpected character after ``x``); it is
40
+ this refusal now, and ``to_latex`` never writes one.
41
+ """
42
+
43
+ import re
44
+
45
+ from .msflparser import MSFLParser
46
+ from .naming import ParsingError
47
+
48
+
49
+ class LatexParsingError(ParsingError):
50
+ """A LaTeX input the reader refuses, carrying a plain message.
51
+
52
+ Subclasses :class:`~unicode_logic_kit.fol.naming.ParsingError`, so every
53
+ caller that handles the parser's own failures handles this one too; it takes
54
+ a string instead of a Lark exception, the way the other importers' errors
55
+ do (:class:`~unicode_logic_kit.fol.prover9_input.Prover9ParsingError`).
56
+ """
57
+
58
+ def __init__(self, message: str):
59
+ self.args = (message,)
60
+
61
+ def __str__(self):
62
+ return self.args[0]
63
+
64
+
65
+ # Control sequences whose argument-brace constructs must be resolved before the
66
+ # generic ``[a-zA-Z]+`` control-sequence pass and before brace stripping, since
67
+ # each one literally contains braces. ``\mathbin{\mathsf{U}}`` is listed before
68
+ # the bare ``\mathsf{U}`` would ever be considered, so the Until glyph wins.
69
+ #
70
+ # Every entry here is a full, self-contained literal string, so list order is
71
+ # safe regardless of shared prefixes: e.g. ``\mathsf{Say}`` and ``\mathsf{S}``
72
+ # (the latter never actually registered — Since uses the overlined
73
+ # ``\mathbin{\overline{\mathsf{S}}}`` form below) would not collide even if
74
+ # both were present, because ``str.replace`` matches the full literal, not a
75
+ # prefix. Any ``\mathsf{...}`` construct NOT listed here (a nominal) falls
76
+ # through to the generic unwrap in step 2 of :func:`latex_to_unicode`.
77
+ _MULTI_TOKEN = [
78
+ # Past-tense overlined markers FIRST: each contains a bare \mathsf{…} that a
79
+ # later rule would otherwise rewrite (e.g. \overline{\mathsf{P}} ⊃ \mathsf{P}).
80
+ (r"\mathbin{\overline{\mathsf{S}}}", "⒮"),
81
+ (r"\overline{\mathsf{H}}", "⒣"),
82
+ (r"\overline{\mathsf{P}}", "⒫"),
83
+ (r"\overline{\mathsf{Y}}", "⒴"),
84
+ (r"\mathbin{\mathsf{U}}", "Ⓤ"),
85
+ (r"\mathsf{G}", "Ⓖ"),
86
+ (r"\mathsf{F}", "Ⓕ"),
87
+ (r"\mathsf{X}", "Ⓝ"),
88
+ (r"\mathsf{O}", "Ⓞ"),
89
+ (r"\mathsf{P}", "Ⓟ"),
90
+ # Assertive Say_<agent> / bouletic Want_<agent> (agent_prefix, like K_a/B_a).
91
+ (r"\mathsf{Say}", "Say"),
92
+ (r"\mathsf{Want}", "Want"),
93
+ # Contrast (concessive but/whereas) and the box-arrow / diamond-arrow
94
+ # counterfactual conditionals — all registered "mathbin" multi-token forms.
95
+ (r"\mathbin{\mathsf{C}}", "Ⓒ"),
96
+ (r"\mathbin{\Box\!\rightarrow}", "□→"),
97
+ (r"\mathbin{\Diamond\!\rightarrow}", "◇→"),
98
+ # Linear logic: the exponential "!" and additive "with" (&) markup.
99
+ (r"\mathord{!}", "!"),
100
+ (r"\mathbin{\&}", "&"),
101
+ # Linear logic multiplicative unit.
102
+ (r"\mathbf{1}", "𝟙"),
103
+ (r"\left(", "("),
104
+ (r"\right)", ")"),
105
+ (r"\left[", "("),
106
+ (r"\right]", ")"),
107
+ (r"\left{", "("),
108
+ (r"\right}", ")"),
109
+ (r"{:}", ":"),
110
+ ]
111
+
112
+ # Backslash control sequences mapped glyph-for-glyph. Keys are the bare names
113
+ # (the ``[a-zA-Z]+`` run after the backslash); lookup is by the FULL run, so a
114
+ # longer name (``leftrightarrow``) is never shadowed by a prefix (``leq`` vs
115
+ # ``le``). Both the glyphs emitted by ``to_latex`` and the common hand-written
116
+ # synonyms are included.
117
+ _CONTROL_SEQUENCES = {
118
+ # Quantifiers.
119
+ "forall": "∀",
120
+ "exists": "∃",
121
+ # Negation.
122
+ "lnot": "¬",
123
+ "neg": "¬",
124
+ # Conjunction / disjunction.
125
+ "land": "∧",
126
+ "wedge": "∧",
127
+ "lor": "∨",
128
+ "vee": "∨",
129
+ # Łukasiewicz strong connectives / linear-logic tensor & plus (shared glyphs).
130
+ "otimes": "⊗",
131
+ "oplus": "⊕",
132
+ # Implication / equivalence.
133
+ "rightarrow": "→",
134
+ "to": "→",
135
+ "implies": "→",
136
+ "leftrightarrow": "↔",
137
+ "iff": "↔",
138
+ # Comparisons.
139
+ "neq": "≠",
140
+ "ne": "≠",
141
+ "leq": "≤",
142
+ "le": "≤",
143
+ "geq": "≥",
144
+ "ge": "≥",
145
+ # Arithmetic.
146
+ "cdot": "*",
147
+ "times": "*",
148
+ # Lambda / degree-measure term.
149
+ "lambda": "λ",
150
+ "mu": "μ",
151
+ # The truth constants (and, in the linear mode, the additive unit ``⊤``).
152
+ "top": "⊤",
153
+ "bot": "⊥",
154
+ # Modal / temporal / deontic prefix operators.
155
+ "Box": "□",
156
+ "Diamond": "◇",
157
+ # Set-cardinality delimiters |{...}|.
158
+ "lvert": "|",
159
+ "rvert": "|",
160
+ # Linear implication (lollipop) and the Lambek product.
161
+ "multimap": "⊸",
162
+ "bullet": "•",
163
+ # Lambek "under" \: the LITERAL backslash character (a formula-level
164
+ # connective in the lambek mode, not LaTeX escaping).
165
+ "backslash": "\\",
166
+ }
167
+
168
+ # LaTeX spacing macros built from a backslash plus a NON-letter (``\,`` ``\;``
169
+ # ``\!`` and a literal backslash-space). These are deleted outright. The
170
+ # letter-run spacing macros ``\quad`` / ``\qquad`` are handled by the
171
+ # control-sequence pass (they map to nothing) so they never reach here.
172
+ _SPACING_NONLETTER = re.compile(r"\\[,;!\s]")
173
+
174
+ # ``\mathrm{Sort}`` -> ``Sort``: an unwrapping of the upright-roman sort marker.
175
+ _MATHRM = re.compile(r"\\mathrm\{([^{}]*)\}")
176
+
177
+ # A bare ``\mathsf{name}`` left after the specific _MULTI_TOKEN entries above
178
+ # have consumed every KNOWN operator markup (Say/Want/G/F/X/O/P and the
179
+ # mathbin-wrapped U/C/S forms) denotes a hybrid-logic nominal
180
+ # (``Nominal.to_latex`` renders as ``\mathsf{i}``) — or, defensively, any other
181
+ # hand-written ``\mathsf{...}`` wrapping, which is unwrapped the same way
182
+ # ``\mathrm{Sort}`` is. Run AFTER the _MULTI_TOKEN loop so the operator forms
183
+ # are already gone, and BEFORE the generic control-sequence pass (which would
184
+ # otherwise treat the bare "\mathsf" as an unknown, unmapped control sequence
185
+ # and mangle the result into "mathsfi", silently misreading the nominal).
186
+ _MATHSF = re.compile(r"\\mathsf\{([^{}]*)\}")
187
+
188
+ # Generic subscript braces ``X_{...}`` -> ``X_...``. Covers epistemic/doxastic
189
+ # operators (``K_{alice}`` -> ``K_alice``) and any other braced subscript. The
190
+ # inner group forbids nested braces, which never occur in to_latex subscripts.
191
+ _SUBSCRIPT_BRACES = re.compile(r"_\{([^{}]*)\}")
192
+
193
+ # The hybrid-logic satisfaction operator's ATNOM terminal is ``/@[a-z][a-zA-Z0-9]*/``
194
+ # — glued directly to the nominal, no underscore — even though @ is rendered as
195
+ # a regular agent_prefix operator (like K_a / B_a) and so emits ``@_{i}`` ->
196
+ # (after the subscript-brace pass) ``@_i``. Unlike K_a/B_a/Say_a/Want_a, whose
197
+ # underscore IS part of the grammar terminal, @'s underscore is a renderer
198
+ # artefact only and must be dropped. Matches ONLY "@_", never touching the
199
+ # agent operators' own underscore-bearing terminals.
200
+ _AT_UNDERSCORE = re.compile(r"@_([a-zA-Z][a-zA-Z0-9]*)")
201
+
202
+ # The counting quantifier's LaTeX form ``\exists^{\geq 3}`` / ``\exists^{\leq
203
+ # n}`` / ``\exists^{= n}`` (Count.to_latex / SortedCount.to_latex) must collapse
204
+ # to the single glued COUNTOP+NUMBER terminal ``∃≥3`` / ``∃≤n`` / ``∃=n`` the
205
+ # grammar expects — brace-stripping alone would leave a stray "^" and a space
206
+ # between the relation and the bound, neither of which the COUNTOP terminal
207
+ # (``/∃[≥≤=]/``) tolerates. Runs AFTER the control-sequence pass (so \geq/\leq
208
+ # have already become ≥/≤) and BEFORE brace-stripping (so the ``{...}``
209
+ # boundary is still there to anchor the match).
210
+ _COUNT_EXPONENT = re.compile(r"∃\^\{\s*([≥≤=])\s*(\d+)\s*\}")
211
+
212
+ # A backslash control sequence: backslash then the LONGEST run of letters.
213
+ _CONTROL_SEQ = re.compile(r"\\([a-zA-Z]+)")
214
+
215
+ # The letter-run spacing macros, mapped to empty so the control-sequence pass
216
+ # deletes them. Kept separate from _CONTROL_SEQUENCES (which holds real glyphs)
217
+ # purely for readability.
218
+ _SPACING_LETTER = {"quad": "", "qquad": ""}
219
+
220
+ # Placeholders protecting escaped literal braces (``\{`` ``\}``) — used by the
221
+ # cardinality term ``\lvert\{v : φ\}\rvert`` and the slashed existential
222
+ # ``\exists x / \{y, z\}\, φ`` — from the generic brace-stripping step, which
223
+ # must remove ordinary LaTeX GROUPING braces but leave these literal ones
224
+ # behind (the Unicode surface syntax for both constructs uses real ``{`` ``}``
225
+ # characters). Private-use-area code points: they cannot occur in any LaTeX
226
+ # input this translator is meant to accept, and no pipeline regex below
227
+ # matches them, so once step 3 substitutes them in, they pass through steps
228
+ # 4-8 inertly until step 9 restores them right after the brace strip.
229
+ _LBRACE_PLACEHOLDER = ""
230
+ _RBRACE_PLACEHOLDER = ""
231
+
232
+
233
+ def _replace_control_seq(match: "re.Match") -> str:
234
+ """Map one backslash control sequence to its Unicode glyph (or to nothing).
235
+
236
+ The full letter run is looked up so that, e.g., ``\\leftrightarrow`` resolves
237
+ as a whole and is never mis-split into ``\\le`` + ``ftrightarrow``. A spacing
238
+ macro (``\\quad`` / ``\\qquad``) maps to the empty string. An unknown control
239
+ sequence is left verbatim (minus the backslash) so the downstream parser can
240
+ surface a precise error rather than this translator swallowing it.
241
+ """
242
+ name = match.group(1)
243
+ if name in _CONTROL_SEQUENCES:
244
+ return _CONTROL_SEQUENCES[name]
245
+ if name in _SPACING_LETTER:
246
+ return _SPACING_LETTER[name]
247
+ return name
248
+
249
+
250
+ def _refuse_quote(text: str) -> None:
251
+ """Raise a :class:`LatexParsingError` when ``text`` holds a single quote.
252
+
253
+ The refusal comes before any substitution, because the substitutions are
254
+ what would change a quoted name. It names the first quote by its position in
255
+ ``text`` (counted from 1, like the positions of the parser's messages).
256
+ """
257
+ index = text.find("'")
258
+ if index < 0:
259
+ return
260
+ raise LatexParsingError(
261
+ "SYNTAX_ERROR: the LaTeX reader does not read a quoted constant ('k2'): "
262
+ f"the quote at position {index + 1} would start one, and the "
263
+ "substitutions of this reader (spaces collapsed, braces removed, \\_ "
264
+ "turned into _) would change the name between the quotes. Write the "
265
+ "formula in the Unicode syntax instead, where 'k2' is the constant "
266
+ "named k2. A prime (x') is refused for the same reason.")
267
+
268
+
269
+ def latex_to_unicode(text: str) -> str:
270
+ """Translate a LaTeX math-mode formula into the toolkit's Unicode surface syntax.
271
+
272
+ The result is a Unicode string ready for :class:`MSFLParser`. A text that
273
+ holds a single quote is refused (see below). The pipeline:
274
+
275
+ 1. Resolve multi-token brace constructs (``\\mathbin{\\mathsf{U}}``,
276
+ ``\\mathsf{G}`` and the other temporal/deontic/agentive markers,
277
+ ``\\mathord{!}``, ``\\mathbin{\\&}``, ``\\mathbf{1}``, ``\\left(`` /
278
+ ``\\right)`` grouping, ``{:}`` the sort colon).
279
+ 2. Unwrap any REMAINING ``\\mathsf{name}`` (a hybrid-logic nominal — every
280
+ KNOWN ``\\mathsf{...}`` operator form was already consumed in step 1).
281
+ 3. Protect escaped literal braces ``\\{`` / ``\\}`` (cardinality terms,
282
+ slashed existentials) behind placeholders so step 9 does not erase them.
283
+ 4. Unescape ``\\_`` to a literal underscore (the ``c_``-constant escape that
284
+ ``to_latex`` emits) so it is not later read as a subscript operator.
285
+ 5. Unwrap ``\\mathrm{Sort}`` to ``Sort``.
286
+ 6. Collapse generic subscript braces ``X_{...}`` to ``X_...``, then tighten
287
+ the hybrid satisfaction operator's ``@_i`` to ``@i`` (its underscore is a
288
+ renderer artefact, unlike the agent operators' K_a/B_a/Say_a/Want_a).
289
+ 7. Delete the non-letter spacing macros (``\\,`` ``\\;`` ``\\!`` and a
290
+ backslash-space) BEFORE mapping control sequences: the ``backslash``
291
+ control sequence (Lambek's *under* connective) maps to a literal
292
+ backslash character, and if spacing deletion ran afterwards it would
293
+ mistake that freshly-produced backslash — now followed by a plain
294
+ space, e.g. ``A \\ B`` — for a backslash-space spacing macro and erase
295
+ it, silently dropping the connective.
296
+ 8. Map every remaining backslash control sequence by its full letter run
297
+ (longest-match), covering both the ``to_latex`` glyphs and common
298
+ hand-written synonyms; the letter-run spacing macros map to nothing.
299
+ 9. Collapse the counting quantifier's exponent (``\\exists^{\\geq 3}`` ->
300
+ ``∃≥3``), then strip leftover grouping braces ``{`` ``}`` (LaTeX
301
+ grouping carries no information the parser needs) and restore the
302
+ placeholders from step 3 to real literal braces.
303
+ 10. Collapse redundant whitespace.
304
+
305
+ Steps 5 to 10 rewrite text without knowing where a quoted name begins and
306
+ ends, so a text that holds a single quote is refused up front instead of
307
+ being rewritten: the Unicode syntax reads ``'k2'`` and ``'John Doe'`` as
308
+ constants named exactly that, and these steps would change the name. This
309
+ includes a prime (``x'``), which was a syntax error of the parser before
310
+ and is this refusal now.
311
+
312
+ Args:
313
+ text: a LaTeX math-mode formula.
314
+
315
+ Returns:
316
+ The same formula in the Unicode surface syntax.
317
+
318
+ Raises:
319
+ LatexParsingError: ``text`` holds a single quote. The message says that
320
+ the LaTeX reader does not read a quoted constant and that the
321
+ formula is to be written in the Unicode syntax.
322
+ """
323
+ _refuse_quote(text)
324
+ s = text
325
+
326
+ # 1. Multi-token brace constructs, most specific first.
327
+ for src, dst in _MULTI_TOKEN:
328
+ s = s.replace(src, dst)
329
+
330
+ # 2. Any \mathsf{...} surviving step 1 is a nominal (or an unrecognised
331
+ # hand-written \mathsf wrapping) — unwrap it to its bare name.
332
+ s = _MATHSF.sub(r"\1", s)
333
+
334
+ # 3. Protect escaped literal braces from the brace-strip in step 9.
335
+ s = s.replace("\\{", _LBRACE_PLACEHOLDER).replace("\\}", _RBRACE_PLACEHOLDER)
336
+
337
+ # 4. Unescape the c_-constant underscore escape (\_ -> _). Done before the
338
+ # subscript-brace and control-sequence passes so the bare underscore in a
339
+ # name like c_zero survives intact.
340
+ s = s.replace("\\_", "_")
341
+
342
+ # 5. Unwrap \mathrm{Sort}.
343
+ s = _MATHRM.sub(r"\1", s)
344
+
345
+ # 6. Generic subscript braces X_{...} -> X_... then @_i -> @i.
346
+ s = _SUBSCRIPT_BRACES.sub(r"_\1", s)
347
+ s = _AT_UNDERSCORE.sub(r"@\1", s)
348
+
349
+ # 7. Non-letter LaTeX spacing macros — deliberately BEFORE control
350
+ # sequences (see the docstring above: a literal backslash produced by
351
+ # step 8's "backslash" mapping must not be mistaken for a spacing macro).
352
+ s = _SPACING_NONLETTER.sub(" ", s)
353
+
354
+ # 8. Backslash control sequences (longest letter run wins).
355
+ s = _CONTROL_SEQ.sub(_replace_control_seq, s)
356
+
357
+ # 9. Counting-quantifier exponent, then leftover-brace stripping / restore.
358
+ s = _COUNT_EXPONENT.sub(lambda m: f"∃{m.group(1)}{m.group(2)}", s)
359
+ s = s.replace("{", "").replace("}", "")
360
+ s = s.replace(_LBRACE_PLACEHOLDER, "{").replace(_RBRACE_PLACEHOLDER, "}")
361
+
362
+ # 10. Collapse redundant whitespace.
363
+ s = re.sub(r"\s+", " ", s).strip()
364
+
365
+ # 11. Tighten the sort colon: the grammar's SORT terminal is /:[A-Z][...]*/ ,
366
+ # which admits no whitespace before the ':' or between ':' and the sort
367
+ # name. ``to_latex`` emits the colon glued (``x{:}\mathrm{Human}``), but a
368
+ # person may hand-write ``x {:} \mathrm{Human}`` or ``x : Human``; the
369
+ # spaces introduced by steps 1/8/10 would then split the SORT token. Since
370
+ # ':' occurs nowhere else in any grammar, removing whitespace flanking it
371
+ # is unambiguous and only ever reconstructs a sort annotation.
372
+ s = re.sub(r"\s*:\s*", ":", s)
373
+ return s
374
+
375
+
376
+ def parse_latex(text: str, many_sorted: bool = False, fuzzy: bool = False,
377
+ modal: bool = False, second_order: bool = False,
378
+ dependence: bool = False, linear: bool = False,
379
+ lambek: bool = False) -> "object":
380
+ """Parse a LaTeX math-mode formula into an AST node.
381
+
382
+ Translates ``text`` to the toolkit's Unicode surface syntax with
383
+ :func:`latex_to_unicode`, then parses it with :class:`MSFLParser` in the
384
+ selected mode. The mode flags are passed straight through to
385
+ :class:`MSFLParser`, including its mutual-exclusivity rules (e.g.
386
+ ``dependence=True`` cannot be combined with any other flag).
387
+
388
+ The Unicode surface syntax produced by the translation must be valid for the
389
+ chosen mode: e.g. modal operators require ``modal=True``, sort annotations
390
+ require ``many_sorted=True``, and Łukasiewicz strong connectives (⊗ ⊕)
391
+ require ``fuzzy=True``. Mismatched flags surface as the parser's usual
392
+ NamingError / ParsingError.
393
+
394
+ Args:
395
+ text: a LaTeX math-mode formula (no surrounding ``$…$`` needed).
396
+ many_sorted: parse in MSFOL/MSFL mode (sorted quantifiers/constants).
397
+ fuzzy: parse with Łukasiewicz operators.
398
+ modal: parse classical unsorted FOL plus modal/temporal/deontic operators.
399
+ second_order: parse classical unsorted FOL plus second-order quantifiers.
400
+ dependence: parse the team-semantic dependence/IF fragment. Standalone.
401
+ linear: parse propositional intuitionistic linear logic. Standalone.
402
+ lambek: parse Lambek-calculus category types. Standalone.
403
+
404
+ Returns:
405
+ The parsed AST :class:`~unicode_logic_kit.fol.nodes.Node`.
406
+
407
+ Raises:
408
+ LatexParsingError: ``text`` holds a single quote (a quoted constant such
409
+ as ``'k2'``, or a prime ``x'``): this reader does not read quotes,
410
+ see :func:`latex_to_unicode`. ``to_latex`` writes a constant by its
411
+ name and never in quotes, so the LaTeX text of a formula that holds
412
+ a constant such as ``k2`` or ``Alice`` does not read back as that
413
+ constant; write such a formula in the Unicode syntax.
414
+ ~unicode_logic_kit.fol.naming.NamingError:
415
+ a name of the translated text is not legal in the chosen mode.
416
+ ~unicode_logic_kit.fol.naming.ParsingError:
417
+ the translated text is not a formula of the chosen mode.
418
+ """
419
+ unicode_text = latex_to_unicode(text)
420
+ parser = MSFLParser(
421
+ many_sorted=many_sorted,
422
+ fuzzy=fuzzy,
423
+ modal=modal,
424
+ second_order=second_order,
425
+ dependence=dependence,
426
+ linear=linear,
427
+ lambek=lambek,
428
+ )
429
+ return parser.parse(unicode_text)