unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,429 @@
|
|
|
1
|
+
"""LaTeX input: read a LaTeX math-mode formula and parse it with MSFLParser.
|
|
2
|
+
|
|
3
|
+
This module is the exact inverse of ``node.to_latex()`` (the LaTeX renderer in
|
|
4
|
+
``_msfl_nodes.py``). It translates LaTeX math-mode markup into the toolkit's
|
|
5
|
+
Unicode surface syntax and then hands the result to :class:`MSFLParser`.
|
|
6
|
+
|
|
7
|
+
The translation is a tokenizing replacement, run as a fixed pipeline (see
|
|
8
|
+
:func:`latex_to_unicode` for the exact, numbered steps): multi-token brace
|
|
9
|
+
constructs (``\\mathbin{\\mathsf{U}}``, ``\\mathsf{G}``, ``\\mathrm{...}``,
|
|
10
|
+
``{:}``, ``\\left(`` / ``\\right)``, subscript braces ``X_{...}``) are
|
|
11
|
+
resolved first; escaped literal braces (``\\{`` ``\\}``, used by the
|
|
12
|
+
cardinality and slashed-existential constructs) are protected from the later
|
|
13
|
+
brace-stripping step; a bare ``\\mathsf{name}`` left over after the specific
|
|
14
|
+
multi-token constructs is unwrapped as a hybrid-logic nominal; then backslash
|
|
15
|
+
control sequences (``\\leftrightarrow``, ``\\forall``, …) are mapped
|
|
16
|
+
glyph-for-glyph by matching the FULL ``[a-zA-Z]+`` run after the backslash (so
|
|
17
|
+
``\\leq`` is never shadowed by ``\\le``); then LaTeX spacing (``\\,`` ``\\;``
|
|
18
|
+
``\\!`` ``\\quad`` ``\\qquad`` and a backslash-space) is deleted; the counting
|
|
19
|
+
quantifier's exponent (``\\exists^{\\geq 3}``) is collapsed to the glued
|
|
20
|
+
``∃≥3`` terminal; finally leftover grouping braces are stripped (the protected
|
|
21
|
+
literal ones are restored right after), since operator precedence in the
|
|
22
|
+
Unicode surface syntax is explicit and LaTeX grouping carries no information
|
|
23
|
+
the parser needs.
|
|
24
|
+
|
|
25
|
+
The control-sequence map also accepts the common hand-written synonyms a person
|
|
26
|
+
would type by hand (``\\neg`` for ``¬``, ``\\to`` for ``→``, ``\\iff`` for
|
|
27
|
+
``↔``, ``\\le`` for ``≤``, ``\\times`` for ``*``, …) so that pasted LaTeX need
|
|
28
|
+
not have come from ``to_latex``.
|
|
29
|
+
|
|
30
|
+
A single quote is the one character this reader refuses. The Unicode syntax
|
|
31
|
+
writes a constant whose name is not a bare word in single quotes (``'k2'``,
|
|
32
|
+
``'John Doe'``), and the name between the quotes is the name exactly; the
|
|
33
|
+
substitutions above know nothing of quotes, so they would collapse the spaces
|
|
34
|
+
of ``'John Doe'``, turn ``'a\\_b'`` into ``'a_b'`` and strip the braces of
|
|
35
|
+
``'f{x}'`` without a word, and the formula would come back with another
|
|
36
|
+
constant. Until this reader tracks quotes it reads no text that holds one: a
|
|
37
|
+
:class:`LatexParsingError` names the quote and says to write the formula in the
|
|
38
|
+
Unicode syntax. A prime (``x'``) is such a quote. Before quoted constants it
|
|
39
|
+
was a syntax error of the parser (an unexpected character after ``x``); it is
|
|
40
|
+
this refusal now, and ``to_latex`` never writes one.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
import re
|
|
44
|
+
|
|
45
|
+
from .msflparser import MSFLParser
|
|
46
|
+
from .naming import ParsingError
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class LatexParsingError(ParsingError):
|
|
50
|
+
"""A LaTeX input the reader refuses, carrying a plain message.
|
|
51
|
+
|
|
52
|
+
Subclasses :class:`~unicode_logic_kit.fol.naming.ParsingError`, so every
|
|
53
|
+
caller that handles the parser's own failures handles this one too; it takes
|
|
54
|
+
a string instead of a Lark exception, the way the other importers' errors
|
|
55
|
+
do (:class:`~unicode_logic_kit.fol.prover9_input.Prover9ParsingError`).
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
def __init__(self, message: str):
|
|
59
|
+
self.args = (message,)
|
|
60
|
+
|
|
61
|
+
def __str__(self):
|
|
62
|
+
return self.args[0]
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
# Control sequences whose argument-brace constructs must be resolved before the
|
|
66
|
+
# generic ``[a-zA-Z]+`` control-sequence pass and before brace stripping, since
|
|
67
|
+
# each one literally contains braces. ``\mathbin{\mathsf{U}}`` is listed before
|
|
68
|
+
# the bare ``\mathsf{U}`` would ever be considered, so the Until glyph wins.
|
|
69
|
+
#
|
|
70
|
+
# Every entry here is a full, self-contained literal string, so list order is
|
|
71
|
+
# safe regardless of shared prefixes: e.g. ``\mathsf{Say}`` and ``\mathsf{S}``
|
|
72
|
+
# (the latter never actually registered — Since uses the overlined
|
|
73
|
+
# ``\mathbin{\overline{\mathsf{S}}}`` form below) would not collide even if
|
|
74
|
+
# both were present, because ``str.replace`` matches the full literal, not a
|
|
75
|
+
# prefix. Any ``\mathsf{...}`` construct NOT listed here (a nominal) falls
|
|
76
|
+
# through to the generic unwrap in step 2 of :func:`latex_to_unicode`.
|
|
77
|
+
_MULTI_TOKEN = [
|
|
78
|
+
# Past-tense overlined markers FIRST: each contains a bare \mathsf{…} that a
|
|
79
|
+
# later rule would otherwise rewrite (e.g. \overline{\mathsf{P}} ⊃ \mathsf{P}).
|
|
80
|
+
(r"\mathbin{\overline{\mathsf{S}}}", "⒮"),
|
|
81
|
+
(r"\overline{\mathsf{H}}", "⒣"),
|
|
82
|
+
(r"\overline{\mathsf{P}}", "⒫"),
|
|
83
|
+
(r"\overline{\mathsf{Y}}", "⒴"),
|
|
84
|
+
(r"\mathbin{\mathsf{U}}", "Ⓤ"),
|
|
85
|
+
(r"\mathsf{G}", "Ⓖ"),
|
|
86
|
+
(r"\mathsf{F}", "Ⓕ"),
|
|
87
|
+
(r"\mathsf{X}", "Ⓝ"),
|
|
88
|
+
(r"\mathsf{O}", "Ⓞ"),
|
|
89
|
+
(r"\mathsf{P}", "Ⓟ"),
|
|
90
|
+
# Assertive Say_<agent> / bouletic Want_<agent> (agent_prefix, like K_a/B_a).
|
|
91
|
+
(r"\mathsf{Say}", "Say"),
|
|
92
|
+
(r"\mathsf{Want}", "Want"),
|
|
93
|
+
# Contrast (concessive but/whereas) and the box-arrow / diamond-arrow
|
|
94
|
+
# counterfactual conditionals — all registered "mathbin" multi-token forms.
|
|
95
|
+
(r"\mathbin{\mathsf{C}}", "Ⓒ"),
|
|
96
|
+
(r"\mathbin{\Box\!\rightarrow}", "□→"),
|
|
97
|
+
(r"\mathbin{\Diamond\!\rightarrow}", "◇→"),
|
|
98
|
+
# Linear logic: the exponential "!" and additive "with" (&) markup.
|
|
99
|
+
(r"\mathord{!}", "!"),
|
|
100
|
+
(r"\mathbin{\&}", "&"),
|
|
101
|
+
# Linear logic multiplicative unit.
|
|
102
|
+
(r"\mathbf{1}", "𝟙"),
|
|
103
|
+
(r"\left(", "("),
|
|
104
|
+
(r"\right)", ")"),
|
|
105
|
+
(r"\left[", "("),
|
|
106
|
+
(r"\right]", ")"),
|
|
107
|
+
(r"\left{", "("),
|
|
108
|
+
(r"\right}", ")"),
|
|
109
|
+
(r"{:}", ":"),
|
|
110
|
+
]
|
|
111
|
+
|
|
112
|
+
# Backslash control sequences mapped glyph-for-glyph. Keys are the bare names
|
|
113
|
+
# (the ``[a-zA-Z]+`` run after the backslash); lookup is by the FULL run, so a
|
|
114
|
+
# longer name (``leftrightarrow``) is never shadowed by a prefix (``leq`` vs
|
|
115
|
+
# ``le``). Both the glyphs emitted by ``to_latex`` and the common hand-written
|
|
116
|
+
# synonyms are included.
|
|
117
|
+
_CONTROL_SEQUENCES = {
|
|
118
|
+
# Quantifiers.
|
|
119
|
+
"forall": "∀",
|
|
120
|
+
"exists": "∃",
|
|
121
|
+
# Negation.
|
|
122
|
+
"lnot": "¬",
|
|
123
|
+
"neg": "¬",
|
|
124
|
+
# Conjunction / disjunction.
|
|
125
|
+
"land": "∧",
|
|
126
|
+
"wedge": "∧",
|
|
127
|
+
"lor": "∨",
|
|
128
|
+
"vee": "∨",
|
|
129
|
+
# Łukasiewicz strong connectives / linear-logic tensor & plus (shared glyphs).
|
|
130
|
+
"otimes": "⊗",
|
|
131
|
+
"oplus": "⊕",
|
|
132
|
+
# Implication / equivalence.
|
|
133
|
+
"rightarrow": "→",
|
|
134
|
+
"to": "→",
|
|
135
|
+
"implies": "→",
|
|
136
|
+
"leftrightarrow": "↔",
|
|
137
|
+
"iff": "↔",
|
|
138
|
+
# Comparisons.
|
|
139
|
+
"neq": "≠",
|
|
140
|
+
"ne": "≠",
|
|
141
|
+
"leq": "≤",
|
|
142
|
+
"le": "≤",
|
|
143
|
+
"geq": "≥",
|
|
144
|
+
"ge": "≥",
|
|
145
|
+
# Arithmetic.
|
|
146
|
+
"cdot": "*",
|
|
147
|
+
"times": "*",
|
|
148
|
+
# Lambda / degree-measure term.
|
|
149
|
+
"lambda": "λ",
|
|
150
|
+
"mu": "μ",
|
|
151
|
+
# The truth constants (and, in the linear mode, the additive unit ``⊤``).
|
|
152
|
+
"top": "⊤",
|
|
153
|
+
"bot": "⊥",
|
|
154
|
+
# Modal / temporal / deontic prefix operators.
|
|
155
|
+
"Box": "□",
|
|
156
|
+
"Diamond": "◇",
|
|
157
|
+
# Set-cardinality delimiters |{...}|.
|
|
158
|
+
"lvert": "|",
|
|
159
|
+
"rvert": "|",
|
|
160
|
+
# Linear implication (lollipop) and the Lambek product.
|
|
161
|
+
"multimap": "⊸",
|
|
162
|
+
"bullet": "•",
|
|
163
|
+
# Lambek "under" \: the LITERAL backslash character (a formula-level
|
|
164
|
+
# connective in the lambek mode, not LaTeX escaping).
|
|
165
|
+
"backslash": "\\",
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
# LaTeX spacing macros built from a backslash plus a NON-letter (``\,`` ``\;``
|
|
169
|
+
# ``\!`` and a literal backslash-space). These are deleted outright. The
|
|
170
|
+
# letter-run spacing macros ``\quad`` / ``\qquad`` are handled by the
|
|
171
|
+
# control-sequence pass (they map to nothing) so they never reach here.
|
|
172
|
+
_SPACING_NONLETTER = re.compile(r"\\[,;!\s]")
|
|
173
|
+
|
|
174
|
+
# ``\mathrm{Sort}`` -> ``Sort``: an unwrapping of the upright-roman sort marker.
|
|
175
|
+
_MATHRM = re.compile(r"\\mathrm\{([^{}]*)\}")
|
|
176
|
+
|
|
177
|
+
# A bare ``\mathsf{name}`` left after the specific _MULTI_TOKEN entries above
|
|
178
|
+
# have consumed every KNOWN operator markup (Say/Want/G/F/X/O/P and the
|
|
179
|
+
# mathbin-wrapped U/C/S forms) denotes a hybrid-logic nominal
|
|
180
|
+
# (``Nominal.to_latex`` renders as ``\mathsf{i}``) — or, defensively, any other
|
|
181
|
+
# hand-written ``\mathsf{...}`` wrapping, which is unwrapped the same way
|
|
182
|
+
# ``\mathrm{Sort}`` is. Run AFTER the _MULTI_TOKEN loop so the operator forms
|
|
183
|
+
# are already gone, and BEFORE the generic control-sequence pass (which would
|
|
184
|
+
# otherwise treat the bare "\mathsf" as an unknown, unmapped control sequence
|
|
185
|
+
# and mangle the result into "mathsfi", silently misreading the nominal).
|
|
186
|
+
_MATHSF = re.compile(r"\\mathsf\{([^{}]*)\}")
|
|
187
|
+
|
|
188
|
+
# Generic subscript braces ``X_{...}`` -> ``X_...``. Covers epistemic/doxastic
|
|
189
|
+
# operators (``K_{alice}`` -> ``K_alice``) and any other braced subscript. The
|
|
190
|
+
# inner group forbids nested braces, which never occur in to_latex subscripts.
|
|
191
|
+
_SUBSCRIPT_BRACES = re.compile(r"_\{([^{}]*)\}")
|
|
192
|
+
|
|
193
|
+
# The hybrid-logic satisfaction operator's ATNOM terminal is ``/@[a-z][a-zA-Z0-9]*/``
|
|
194
|
+
# — glued directly to the nominal, no underscore — even though @ is rendered as
|
|
195
|
+
# a regular agent_prefix operator (like K_a / B_a) and so emits ``@_{i}`` ->
|
|
196
|
+
# (after the subscript-brace pass) ``@_i``. Unlike K_a/B_a/Say_a/Want_a, whose
|
|
197
|
+
# underscore IS part of the grammar terminal, @'s underscore is a renderer
|
|
198
|
+
# artefact only and must be dropped. Matches ONLY "@_", never touching the
|
|
199
|
+
# agent operators' own underscore-bearing terminals.
|
|
200
|
+
_AT_UNDERSCORE = re.compile(r"@_([a-zA-Z][a-zA-Z0-9]*)")
|
|
201
|
+
|
|
202
|
+
# The counting quantifier's LaTeX form ``\exists^{\geq 3}`` / ``\exists^{\leq
|
|
203
|
+
# n}`` / ``\exists^{= n}`` (Count.to_latex / SortedCount.to_latex) must collapse
|
|
204
|
+
# to the single glued COUNTOP+NUMBER terminal ``∃≥3`` / ``∃≤n`` / ``∃=n`` the
|
|
205
|
+
# grammar expects — brace-stripping alone would leave a stray "^" and a space
|
|
206
|
+
# between the relation and the bound, neither of which the COUNTOP terminal
|
|
207
|
+
# (``/∃[≥≤=]/``) tolerates. Runs AFTER the control-sequence pass (so \geq/\leq
|
|
208
|
+
# have already become ≥/≤) and BEFORE brace-stripping (so the ``{...}``
|
|
209
|
+
# boundary is still there to anchor the match).
|
|
210
|
+
_COUNT_EXPONENT = re.compile(r"∃\^\{\s*([≥≤=])\s*(\d+)\s*\}")
|
|
211
|
+
|
|
212
|
+
# A backslash control sequence: backslash then the LONGEST run of letters.
|
|
213
|
+
_CONTROL_SEQ = re.compile(r"\\([a-zA-Z]+)")
|
|
214
|
+
|
|
215
|
+
# The letter-run spacing macros, mapped to empty so the control-sequence pass
|
|
216
|
+
# deletes them. Kept separate from _CONTROL_SEQUENCES (which holds real glyphs)
|
|
217
|
+
# purely for readability.
|
|
218
|
+
_SPACING_LETTER = {"quad": "", "qquad": ""}
|
|
219
|
+
|
|
220
|
+
# Placeholders protecting escaped literal braces (``\{`` ``\}``) — used by the
|
|
221
|
+
# cardinality term ``\lvert\{v : φ\}\rvert`` and the slashed existential
|
|
222
|
+
# ``\exists x / \{y, z\}\, φ`` — from the generic brace-stripping step, which
|
|
223
|
+
# must remove ordinary LaTeX GROUPING braces but leave these literal ones
|
|
224
|
+
# behind (the Unicode surface syntax for both constructs uses real ``{`` ``}``
|
|
225
|
+
# characters). Private-use-area code points: they cannot occur in any LaTeX
|
|
226
|
+
# input this translator is meant to accept, and no pipeline regex below
|
|
227
|
+
# matches them, so once step 3 substitutes them in, they pass through steps
|
|
228
|
+
# 4-8 inertly until step 9 restores them right after the brace strip.
|
|
229
|
+
_LBRACE_PLACEHOLDER = ""
|
|
230
|
+
_RBRACE_PLACEHOLDER = ""
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def _replace_control_seq(match: "re.Match") -> str:
|
|
234
|
+
"""Map one backslash control sequence to its Unicode glyph (or to nothing).
|
|
235
|
+
|
|
236
|
+
The full letter run is looked up so that, e.g., ``\\leftrightarrow`` resolves
|
|
237
|
+
as a whole and is never mis-split into ``\\le`` + ``ftrightarrow``. A spacing
|
|
238
|
+
macro (``\\quad`` / ``\\qquad``) maps to the empty string. An unknown control
|
|
239
|
+
sequence is left verbatim (minus the backslash) so the downstream parser can
|
|
240
|
+
surface a precise error rather than this translator swallowing it.
|
|
241
|
+
"""
|
|
242
|
+
name = match.group(1)
|
|
243
|
+
if name in _CONTROL_SEQUENCES:
|
|
244
|
+
return _CONTROL_SEQUENCES[name]
|
|
245
|
+
if name in _SPACING_LETTER:
|
|
246
|
+
return _SPACING_LETTER[name]
|
|
247
|
+
return name
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def _refuse_quote(text: str) -> None:
|
|
251
|
+
"""Raise a :class:`LatexParsingError` when ``text`` holds a single quote.
|
|
252
|
+
|
|
253
|
+
The refusal comes before any substitution, because the substitutions are
|
|
254
|
+
what would change a quoted name. It names the first quote by its position in
|
|
255
|
+
``text`` (counted from 1, like the positions of the parser's messages).
|
|
256
|
+
"""
|
|
257
|
+
index = text.find("'")
|
|
258
|
+
if index < 0:
|
|
259
|
+
return
|
|
260
|
+
raise LatexParsingError(
|
|
261
|
+
"SYNTAX_ERROR: the LaTeX reader does not read a quoted constant ('k2'): "
|
|
262
|
+
f"the quote at position {index + 1} would start one, and the "
|
|
263
|
+
"substitutions of this reader (spaces collapsed, braces removed, \\_ "
|
|
264
|
+
"turned into _) would change the name between the quotes. Write the "
|
|
265
|
+
"formula in the Unicode syntax instead, where 'k2' is the constant "
|
|
266
|
+
"named k2. A prime (x') is refused for the same reason.")
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def latex_to_unicode(text: str) -> str:
|
|
270
|
+
"""Translate a LaTeX math-mode formula into the toolkit's Unicode surface syntax.
|
|
271
|
+
|
|
272
|
+
The result is a Unicode string ready for :class:`MSFLParser`. A text that
|
|
273
|
+
holds a single quote is refused (see below). The pipeline:
|
|
274
|
+
|
|
275
|
+
1. Resolve multi-token brace constructs (``\\mathbin{\\mathsf{U}}``,
|
|
276
|
+
``\\mathsf{G}`` and the other temporal/deontic/agentive markers,
|
|
277
|
+
``\\mathord{!}``, ``\\mathbin{\\&}``, ``\\mathbf{1}``, ``\\left(`` /
|
|
278
|
+
``\\right)`` grouping, ``{:}`` the sort colon).
|
|
279
|
+
2. Unwrap any REMAINING ``\\mathsf{name}`` (a hybrid-logic nominal — every
|
|
280
|
+
KNOWN ``\\mathsf{...}`` operator form was already consumed in step 1).
|
|
281
|
+
3. Protect escaped literal braces ``\\{`` / ``\\}`` (cardinality terms,
|
|
282
|
+
slashed existentials) behind placeholders so step 9 does not erase them.
|
|
283
|
+
4. Unescape ``\\_`` to a literal underscore (the ``c_``-constant escape that
|
|
284
|
+
``to_latex`` emits) so it is not later read as a subscript operator.
|
|
285
|
+
5. Unwrap ``\\mathrm{Sort}`` to ``Sort``.
|
|
286
|
+
6. Collapse generic subscript braces ``X_{...}`` to ``X_...``, then tighten
|
|
287
|
+
the hybrid satisfaction operator's ``@_i`` to ``@i`` (its underscore is a
|
|
288
|
+
renderer artefact, unlike the agent operators' K_a/B_a/Say_a/Want_a).
|
|
289
|
+
7. Delete the non-letter spacing macros (``\\,`` ``\\;`` ``\\!`` and a
|
|
290
|
+
backslash-space) BEFORE mapping control sequences: the ``backslash``
|
|
291
|
+
control sequence (Lambek's *under* connective) maps to a literal
|
|
292
|
+
backslash character, and if spacing deletion ran afterwards it would
|
|
293
|
+
mistake that freshly-produced backslash — now followed by a plain
|
|
294
|
+
space, e.g. ``A \\ B`` — for a backslash-space spacing macro and erase
|
|
295
|
+
it, silently dropping the connective.
|
|
296
|
+
8. Map every remaining backslash control sequence by its full letter run
|
|
297
|
+
(longest-match), covering both the ``to_latex`` glyphs and common
|
|
298
|
+
hand-written synonyms; the letter-run spacing macros map to nothing.
|
|
299
|
+
9. Collapse the counting quantifier's exponent (``\\exists^{\\geq 3}`` ->
|
|
300
|
+
``∃≥3``), then strip leftover grouping braces ``{`` ``}`` (LaTeX
|
|
301
|
+
grouping carries no information the parser needs) and restore the
|
|
302
|
+
placeholders from step 3 to real literal braces.
|
|
303
|
+
10. Collapse redundant whitespace.
|
|
304
|
+
|
|
305
|
+
Steps 5 to 10 rewrite text without knowing where a quoted name begins and
|
|
306
|
+
ends, so a text that holds a single quote is refused up front instead of
|
|
307
|
+
being rewritten: the Unicode syntax reads ``'k2'`` and ``'John Doe'`` as
|
|
308
|
+
constants named exactly that, and these steps would change the name. This
|
|
309
|
+
includes a prime (``x'``), which was a syntax error of the parser before
|
|
310
|
+
and is this refusal now.
|
|
311
|
+
|
|
312
|
+
Args:
|
|
313
|
+
text: a LaTeX math-mode formula.
|
|
314
|
+
|
|
315
|
+
Returns:
|
|
316
|
+
The same formula in the Unicode surface syntax.
|
|
317
|
+
|
|
318
|
+
Raises:
|
|
319
|
+
LatexParsingError: ``text`` holds a single quote. The message says that
|
|
320
|
+
the LaTeX reader does not read a quoted constant and that the
|
|
321
|
+
formula is to be written in the Unicode syntax.
|
|
322
|
+
"""
|
|
323
|
+
_refuse_quote(text)
|
|
324
|
+
s = text
|
|
325
|
+
|
|
326
|
+
# 1. Multi-token brace constructs, most specific first.
|
|
327
|
+
for src, dst in _MULTI_TOKEN:
|
|
328
|
+
s = s.replace(src, dst)
|
|
329
|
+
|
|
330
|
+
# 2. Any \mathsf{...} surviving step 1 is a nominal (or an unrecognised
|
|
331
|
+
# hand-written \mathsf wrapping) — unwrap it to its bare name.
|
|
332
|
+
s = _MATHSF.sub(r"\1", s)
|
|
333
|
+
|
|
334
|
+
# 3. Protect escaped literal braces from the brace-strip in step 9.
|
|
335
|
+
s = s.replace("\\{", _LBRACE_PLACEHOLDER).replace("\\}", _RBRACE_PLACEHOLDER)
|
|
336
|
+
|
|
337
|
+
# 4. Unescape the c_-constant underscore escape (\_ -> _). Done before the
|
|
338
|
+
# subscript-brace and control-sequence passes so the bare underscore in a
|
|
339
|
+
# name like c_zero survives intact.
|
|
340
|
+
s = s.replace("\\_", "_")
|
|
341
|
+
|
|
342
|
+
# 5. Unwrap \mathrm{Sort}.
|
|
343
|
+
s = _MATHRM.sub(r"\1", s)
|
|
344
|
+
|
|
345
|
+
# 6. Generic subscript braces X_{...} -> X_... then @_i -> @i.
|
|
346
|
+
s = _SUBSCRIPT_BRACES.sub(r"_\1", s)
|
|
347
|
+
s = _AT_UNDERSCORE.sub(r"@\1", s)
|
|
348
|
+
|
|
349
|
+
# 7. Non-letter LaTeX spacing macros — deliberately BEFORE control
|
|
350
|
+
# sequences (see the docstring above: a literal backslash produced by
|
|
351
|
+
# step 8's "backslash" mapping must not be mistaken for a spacing macro).
|
|
352
|
+
s = _SPACING_NONLETTER.sub(" ", s)
|
|
353
|
+
|
|
354
|
+
# 8. Backslash control sequences (longest letter run wins).
|
|
355
|
+
s = _CONTROL_SEQ.sub(_replace_control_seq, s)
|
|
356
|
+
|
|
357
|
+
# 9. Counting-quantifier exponent, then leftover-brace stripping / restore.
|
|
358
|
+
s = _COUNT_EXPONENT.sub(lambda m: f"∃{m.group(1)}{m.group(2)}", s)
|
|
359
|
+
s = s.replace("{", "").replace("}", "")
|
|
360
|
+
s = s.replace(_LBRACE_PLACEHOLDER, "{").replace(_RBRACE_PLACEHOLDER, "}")
|
|
361
|
+
|
|
362
|
+
# 10. Collapse redundant whitespace.
|
|
363
|
+
s = re.sub(r"\s+", " ", s).strip()
|
|
364
|
+
|
|
365
|
+
# 11. Tighten the sort colon: the grammar's SORT terminal is /:[A-Z][...]*/ ,
|
|
366
|
+
# which admits no whitespace before the ':' or between ':' and the sort
|
|
367
|
+
# name. ``to_latex`` emits the colon glued (``x{:}\mathrm{Human}``), but a
|
|
368
|
+
# person may hand-write ``x {:} \mathrm{Human}`` or ``x : Human``; the
|
|
369
|
+
# spaces introduced by steps 1/8/10 would then split the SORT token. Since
|
|
370
|
+
# ':' occurs nowhere else in any grammar, removing whitespace flanking it
|
|
371
|
+
# is unambiguous and only ever reconstructs a sort annotation.
|
|
372
|
+
s = re.sub(r"\s*:\s*", ":", s)
|
|
373
|
+
return s
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
def parse_latex(text: str, many_sorted: bool = False, fuzzy: bool = False,
|
|
377
|
+
modal: bool = False, second_order: bool = False,
|
|
378
|
+
dependence: bool = False, linear: bool = False,
|
|
379
|
+
lambek: bool = False) -> "object":
|
|
380
|
+
"""Parse a LaTeX math-mode formula into an AST node.
|
|
381
|
+
|
|
382
|
+
Translates ``text`` to the toolkit's Unicode surface syntax with
|
|
383
|
+
:func:`latex_to_unicode`, then parses it with :class:`MSFLParser` in the
|
|
384
|
+
selected mode. The mode flags are passed straight through to
|
|
385
|
+
:class:`MSFLParser`, including its mutual-exclusivity rules (e.g.
|
|
386
|
+
``dependence=True`` cannot be combined with any other flag).
|
|
387
|
+
|
|
388
|
+
The Unicode surface syntax produced by the translation must be valid for the
|
|
389
|
+
chosen mode: e.g. modal operators require ``modal=True``, sort annotations
|
|
390
|
+
require ``many_sorted=True``, and Łukasiewicz strong connectives (⊗ ⊕)
|
|
391
|
+
require ``fuzzy=True``. Mismatched flags surface as the parser's usual
|
|
392
|
+
NamingError / ParsingError.
|
|
393
|
+
|
|
394
|
+
Args:
|
|
395
|
+
text: a LaTeX math-mode formula (no surrounding ``$…$`` needed).
|
|
396
|
+
many_sorted: parse in MSFOL/MSFL mode (sorted quantifiers/constants).
|
|
397
|
+
fuzzy: parse with Łukasiewicz operators.
|
|
398
|
+
modal: parse classical unsorted FOL plus modal/temporal/deontic operators.
|
|
399
|
+
second_order: parse classical unsorted FOL plus second-order quantifiers.
|
|
400
|
+
dependence: parse the team-semantic dependence/IF fragment. Standalone.
|
|
401
|
+
linear: parse propositional intuitionistic linear logic. Standalone.
|
|
402
|
+
lambek: parse Lambek-calculus category types. Standalone.
|
|
403
|
+
|
|
404
|
+
Returns:
|
|
405
|
+
The parsed AST :class:`~unicode_logic_kit.fol.nodes.Node`.
|
|
406
|
+
|
|
407
|
+
Raises:
|
|
408
|
+
LatexParsingError: ``text`` holds a single quote (a quoted constant such
|
|
409
|
+
as ``'k2'``, or a prime ``x'``): this reader does not read quotes,
|
|
410
|
+
see :func:`latex_to_unicode`. ``to_latex`` writes a constant by its
|
|
411
|
+
name and never in quotes, so the LaTeX text of a formula that holds
|
|
412
|
+
a constant such as ``k2`` or ``Alice`` does not read back as that
|
|
413
|
+
constant; write such a formula in the Unicode syntax.
|
|
414
|
+
~unicode_logic_kit.fol.naming.NamingError:
|
|
415
|
+
a name of the translated text is not legal in the chosen mode.
|
|
416
|
+
~unicode_logic_kit.fol.naming.ParsingError:
|
|
417
|
+
the translated text is not a formula of the chosen mode.
|
|
418
|
+
"""
|
|
419
|
+
unicode_text = latex_to_unicode(text)
|
|
420
|
+
parser = MSFLParser(
|
|
421
|
+
many_sorted=many_sorted,
|
|
422
|
+
fuzzy=fuzzy,
|
|
423
|
+
modal=modal,
|
|
424
|
+
second_order=second_order,
|
|
425
|
+
dependence=dependence,
|
|
426
|
+
linear=linear,
|
|
427
|
+
lambek=lambek,
|
|
428
|
+
)
|
|
429
|
+
return parser.parse(unicode_text)
|