unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,1033 @@
|
|
|
1
|
+
"""MSFLParser: unified parser for FOL, MSFOL, and MSFL modes."""
|
|
2
|
+
|
|
3
|
+
import pathlib
|
|
4
|
+
from lark import Lark, Token, UnexpectedCharacters, UnexpectedToken, UnexpectedEOF
|
|
5
|
+
from lark.exceptions import VisitError
|
|
6
|
+
|
|
7
|
+
from .nodes import Node, FOLTransformer
|
|
8
|
+
from ._msfl_nodes import LambdaVar, Lambda, Application, resolve_lambda_scope
|
|
9
|
+
from ._fol_nodes import (
|
|
10
|
+
Variable, Constant, Function,
|
|
11
|
+
build_grammar, build_transform_handlers, PARSER_OPS,
|
|
12
|
+
OPERATORS, parser_ops_for_mode,
|
|
13
|
+
)
|
|
14
|
+
from ._modal_nodes import resolve_agent_variables
|
|
15
|
+
from ._so_nodes import ConflictingArityError # re-exported for callers/tests
|
|
16
|
+
from ._ho_nodes import ( # PredicateTerm/MixedSlotError re-exported for callers
|
|
17
|
+
PredicateTerm, MixedSlotError, analyse_signatures,
|
|
18
|
+
)
|
|
19
|
+
from .naming import NamingError, ParsingError
|
|
20
|
+
from .spans import (
|
|
21
|
+
SpannedFormula,
|
|
22
|
+
span_from_meta, span_from_token, make_span, find_glyph,
|
|
23
|
+
trim_ws_backward, fill_gap_spans, build_span_map, project_spans,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
_GRAMMARS_DIR = pathlib.Path(__file__).parent / "grammars"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _safe_find_glyph(text: str, start: int, end: int, glyph: str):
|
|
30
|
+
"""``spans.find_glyph``, but degrading to ``None`` instead of raising.
|
|
31
|
+
|
|
32
|
+
Every call site below is guarded by ``spans.find_glyph``'s own
|
|
33
|
+
grammar-guaranteed invariant (the glyph really is the sole non-parens
|
|
34
|
+
content between two adjacent operands) — this should never actually
|
|
35
|
+
fail for a tree the parser itself produced. It is wrapped anyway so a
|
|
36
|
+
span-derivation edge case degrades to that ONE occurrence reporting
|
|
37
|
+
UNKNOWN rather than turning an opt-in, best-effort feature into a hard
|
|
38
|
+
failure of parse_with_spans itself.
|
|
39
|
+
"""
|
|
40
|
+
try:
|
|
41
|
+
return find_glyph(text, start, end, glyph)
|
|
42
|
+
except ValueError:
|
|
43
|
+
return None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
# =========================
|
|
47
|
+
# Single-letter function-call patch
|
|
48
|
+
# =========================
|
|
49
|
+
#
|
|
50
|
+
# The shared (non-registry) term layer in _fol_nodes.py's grammar template only
|
|
51
|
+
# ever lets a NAME (>= 2 letters) head a function call: ``NAME "(" termlist
|
|
52
|
+
# ")" -> function_``. A single lowercase letter lexes as VARIABLE instead of
|
|
53
|
+
# NAME, and ``?atom_term: VARIABLE`` has no continuation into "(", so
|
|
54
|
+
# without the patch below ``f(x)`` fails at the LEXER level (a NamingError: '('
|
|
55
|
+
# is not a valid continuation after VARIABLE in that grammar position) even
|
|
56
|
+
# though ``Function('f', [...])`` is a perfectly legal AST node whose
|
|
57
|
+
# ``to_unicode_str()`` prints ``f(x)``. With the patch that text reads back as
|
|
58
|
+
# the function. (The head of a function stays a bare word: a quoted name is a
|
|
59
|
+
# constant, and ``'f'(x)`` is refused.) Verified unambiguous (no Earley
|
|
60
|
+
# ambiguity against lambda application or the atom/atom_term rules:
|
|
61
|
+
# VARIABLE-as-bare-term and VARIABLE-as-function-head are distinguished purely
|
|
62
|
+
# by whether "(" follows, and a bare term can never itself reduce to a formula,
|
|
63
|
+
# so there is no competing derivation for e.g. "(f)(y)" or
|
|
64
|
+
# "(λx. P(x))(f(y))") by building the patched grammar for every mode and
|
|
65
|
+
# cross-checking against an ``ambiguity="explicit"`` Earley parser, plus
|
|
66
|
+
# running the full parser test suite (test_msfl_parser.py, test_lambda_tools.py,
|
|
67
|
+
# test_resolve_lambda_scope.py).
|
|
68
|
+
#
|
|
69
|
+
# The alternative is applied as a targeted, self-checking patch to the
|
|
70
|
+
# ASSEMBLED grammar string and not written into the shared template
|
|
71
|
+
# (_fol_nodes.py's _BASE_GRAMMAR_TEMPLATE): the patch splices in a
|
|
72
|
+
# ``VARIABLE "(" termlist ")" -> function_`` alternative right next to the
|
|
73
|
+
# existing bare-VARIABLE one, and it finds its place by the head line
|
|
74
|
+
# ``?atom_term: VARIABLE``, so the template keeps that line exactly as it is
|
|
75
|
+
# (the quoted constant is one more alternative after it, not a change to it).
|
|
76
|
+
# The patch is applied identically to every mode, since the term layer is
|
|
77
|
+
# verbatim-shared across all of them (build_grammar's per-mode variation is
|
|
78
|
+
# entirely in the formula-operator layers, not atom_term).
|
|
79
|
+
_ATOM_TERM_VARIABLE_MARKER = '?atom_term: VARIABLE\n'
|
|
80
|
+
_ATOM_TERM_FUNCTION_PATCH = (
|
|
81
|
+
_ATOM_TERM_VARIABLE_MARKER
|
|
82
|
+
+ ' | VARIABLE "(" termlist ")" -> function_\n'
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _allow_single_letter_function_calls(grammar_text: str) -> str:
|
|
87
|
+
"""Splice a VARIABLE-headed function-call alternative into atom_term.
|
|
88
|
+
|
|
89
|
+
Returns ``grammar_text`` with exactly one occurrence of the bare
|
|
90
|
+
``?atom_term: VARIABLE`` line followed by a new
|
|
91
|
+
``VARIABLE "(" termlist ")" -> function_`` alternative (same rule alias as
|
|
92
|
+
the existing NAME-headed case, so no new Transformer method name is
|
|
93
|
+
needed — only ``function_`` itself is extended, see
|
|
94
|
+
:meth:`LambdaTransformer.function_`).
|
|
95
|
+
|
|
96
|
+
Raises RuntimeError if the marker is not found exactly once: a template
|
|
97
|
+
change in ``_fol_nodes.py`` would otherwise make this patch silently a
|
|
98
|
+
no-op and quietly resurrect the single-letter-function bug.
|
|
99
|
+
"""
|
|
100
|
+
count = grammar_text.count(_ATOM_TERM_VARIABLE_MARKER)
|
|
101
|
+
if count != 1:
|
|
102
|
+
raise RuntimeError(
|
|
103
|
+
"MSFLParser: expected exactly one '?atom_term: VARIABLE' line in "
|
|
104
|
+
f"the generated grammar, found {count}. The shared term layer in "
|
|
105
|
+
"_fol_nodes.py's grammar template has changed shape; update "
|
|
106
|
+
"_allow_single_letter_function_calls (msflparser.py) to match, or "
|
|
107
|
+
"the single-letter function-call fix silently stops applying."
|
|
108
|
+
)
|
|
109
|
+
return grammar_text.replace(
|
|
110
|
+
_ATOM_TERM_VARIABLE_MARKER, _ATOM_TERM_FUNCTION_PATCH, 1)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
class LambdaTransformer(FOLTransformer):
|
|
114
|
+
"""Extends FOLTransformer with lambda-abstraction and application handlers.
|
|
115
|
+
|
|
116
|
+
All three parser modes inherit these via the class hierarchy, so λ-syntax
|
|
117
|
+
is available in FOL, MSFOL, and MSFL without duplicating the handlers.
|
|
118
|
+
|
|
119
|
+
Lambda/LambdaVar/Application are defined in _msfl_nodes.py (which imports
|
|
120
|
+
from _fol_nodes.py), so this class lives in msflparser.py rather than
|
|
121
|
+
_fol_nodes.py to avoid a circular import.
|
|
122
|
+
"""
|
|
123
|
+
|
|
124
|
+
def pred_arg_(self, items):
|
|
125
|
+
"""Build a PredicateTerm from a PREDICATE token in ARGUMENT position.
|
|
126
|
+
|
|
127
|
+
Only the third-order modes' ``hoarg`` rule reaches this (see
|
|
128
|
+
``_MODE_ATOM_EXTRA`` in _fol_nodes.py); in every other mode a predicate
|
|
129
|
+
name cannot appear as an argument at all. It lives here rather than on
|
|
130
|
+
FOLTransformer for the same reason ``lambda_`` does: PredicateTerm is
|
|
131
|
+
defined downstream of _fol_nodes.py.
|
|
132
|
+
"""
|
|
133
|
+
return PredicateTerm(str(items[0]))
|
|
134
|
+
|
|
135
|
+
def lambda_(self, items):
|
|
136
|
+
# Grammar: LAMBDA (VARIABLE | NAME | PREDICATE) "." formula
|
|
137
|
+
# LAMBDA is a named terminal, so it appears as items[0] (raw Token).
|
|
138
|
+
# "." is a string literal and is filtered out by Lark.
|
|
139
|
+
# items[1] is the parameter, already processed by terminal handlers:
|
|
140
|
+
# VARIABLE → Variable node (via FOLTransformer.VARIABLE)
|
|
141
|
+
# NAME → Constant node (via FOLTransformer.NAME)
|
|
142
|
+
# PREDICATE → raw Token (no terminal handler for PREDICATE)
|
|
143
|
+
# items[2] is the body node.
|
|
144
|
+
param = items[1]
|
|
145
|
+
if isinstance(param, (Variable, Constant)):
|
|
146
|
+
param_name = param.name
|
|
147
|
+
else:
|
|
148
|
+
param_name = str(param) # raw Token for PREDICATE params, e.g. λP. …
|
|
149
|
+
return Lambda(LambdaVar(param_name), items[2])
|
|
150
|
+
|
|
151
|
+
def application_(self, items):
|
|
152
|
+
return Application(items[0], items[1])
|
|
153
|
+
|
|
154
|
+
def function_(self, items):
|
|
155
|
+
"""Transform a function application into a Function node.
|
|
156
|
+
|
|
157
|
+
Extends ``FOLTransformer.function_`` to also accept a single-letter
|
|
158
|
+
VARIABLE head, not just a multi-letter NAME: the
|
|
159
|
+
``VARIABLE "(" termlist ")" -> function_`` alternative spliced into
|
|
160
|
+
atom_term by :func:`_allow_single_letter_function_calls` reduces to
|
|
161
|
+
this SAME rule alias, so both cases arrive here. Lark transforms
|
|
162
|
+
bottom-up, so by the time this fires the head token has already been
|
|
163
|
+
turned into a node by its terminal handler: a Constant (NAME) or a
|
|
164
|
+
Variable (VARIABLE) — never a raw Token — so both are unwrapped via
|
|
165
|
+
``.name``.
|
|
166
|
+
"""
|
|
167
|
+
head = items[0]
|
|
168
|
+
if isinstance(head, (Constant, Variable)):
|
|
169
|
+
name = head.name
|
|
170
|
+
else:
|
|
171
|
+
name = str(head)
|
|
172
|
+
args = items[1:]
|
|
173
|
+
if args and isinstance(args[0], list):
|
|
174
|
+
args = args[0]
|
|
175
|
+
return Function(name, args)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
# The per-mode hand-written transformers (Modal/SecondOrder/MSFOL/MSFL/FL +
|
|
179
|
+
# LukConnectivesMixin) and the six per-mode .lark grammars were retired once the
|
|
180
|
+
# registry pipeline (build_grammar / build_transform_handlers over PARSER_OPS) was
|
|
181
|
+
# verified to reproduce them byte-for-byte. The runtime now assembles every mode from
|
|
182
|
+
# the registry on a shared LambdaTransformer base (see _assemble_transformer); only
|
|
183
|
+
# terminals.lark survives, imported by the generated grammar.
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
# MSFLParser's short mode name (used in NamingError/ParsingError messages) -> the
|
|
187
|
+
# registry mode key consumed by build_grammar / build_transform_handlers.
|
|
188
|
+
_REGISTRY_MODE = {
|
|
189
|
+
"fol": "fol", "msfol": "msfol", "msfl": "msfl", "fl": "fl",
|
|
190
|
+
"modal": "modal", "so": "second_order",
|
|
191
|
+
"to": "third_order", "tomodal": "third_order_modal",
|
|
192
|
+
"modal_sorted": "modal_sorted", "so_sorted": "so_sorted",
|
|
193
|
+
"dependence": "dependence", "linear": "linear", "lambek": "lambek",
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
# Compiled Lark parsers are cached per (registry_mode, registry size, algorithm):
|
|
198
|
+
# building either grammar is the expensive step, and it depends only on the
|
|
199
|
+
# registered ParserOps. Keying on len(PARSER_OPS) rebuilds automatically if an
|
|
200
|
+
# operator is registered at runtime (the registry is otherwise frozen after
|
|
201
|
+
# import). The algorithm is part of the key because a hybrid mode (below)
|
|
202
|
+
# caches BOTH an LALR and an Earley build of the SAME grammar side by side.
|
|
203
|
+
_PARSER_CACHE: dict = {}
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
# Modes that get an LALR-first, Earley-fallback WRAPPER instead of a single
|
|
207
|
+
# parser. Every other mode is parsed with plain LALR, which on the kit's own
|
|
208
|
+
# corpus is 30-50x faster for identical trees, identical accept/reject sets
|
|
209
|
+
# and identical source spans (4815 formula-level span comparisons, zero
|
|
210
|
+
# differences) -- measured mode by mode, not assumed.
|
|
211
|
+
#
|
|
212
|
+
# `modal` is the one that cannot just move to LALR outright, and for a
|
|
213
|
+
# specific reason rather than caution: `atom_term`'s `"(" term ")"`
|
|
214
|
+
# alternative and `?prefix`'s `"(" formula ")"` alternative (feeding the
|
|
215
|
+
# hybrid-logic bare-nominal rule in _hybrid_nodes.py,
|
|
216
|
+
# `?nominal.-1: (NAME | VARIABLE)`) share an LALR state after `"("` plus a
|
|
217
|
+
# bare lowercase NAME/VARIABLE token. The nominal rule's `-1` priority --
|
|
218
|
+
# added for an unrelated lambda-application disambiguation, see
|
|
219
|
+
# _hybrid_nodes.py -- makes LALR resolve that shared state's reduce/reduce
|
|
220
|
+
# conflict toward the TERM reading even where only the formula/nominal
|
|
221
|
+
# reading can lead to a complete parse, so the parser commits to the wrong
|
|
222
|
+
# branch several tokens early and then fails outright a few tokens later:
|
|
223
|
+
# `(q→p)`, `(p∧q)`, `(p→p)` all raise UnexpectedToken under plain LALR, while
|
|
224
|
+
# the same shapes with an uppercase PREDICATE head, e.g. `(P→P)`, succeed
|
|
225
|
+
# (verified empirically, not just asserted -- see
|
|
226
|
+
# tests/test_parser_backend.py's test_modal_still_accepts_what_only_earley_reaches
|
|
227
|
+
# and the soundness differential in tests/test_modal_lalr_fallback.py). Lark
|
|
228
|
+
# offers only Earley and LALR(1), and this is a genuine LALR(1) conflict the
|
|
229
|
+
# tables resolve by priority, not 8 special-cased strings, so eliminating it
|
|
230
|
+
# cleanly would need nontrivial grammar state-splitting with no guaranteed
|
|
231
|
+
# clean answer -- exactly the kind of grammar surgery that risks the
|
|
232
|
+
# silent-narrowing this project refuses to ship.
|
|
233
|
+
#
|
|
234
|
+
# So `modal` keeps Earley as a FALLBACK rather than as the mode's parser:
|
|
235
|
+
# MSFLParser builds both an LALR parser (the fast path, taken -- and the only
|
|
236
|
+
# one taken -- for every input this conflict doesn't touch) and the Earley
|
|
237
|
+
# parser (the previous, exact parser for this language), tries LALR first,
|
|
238
|
+
# and falls back to Earley transparently on ANY of Lark's three failure
|
|
239
|
+
# exceptions (UnexpectedCharacters/UnexpectedToken/UnexpectedEOF -- the
|
|
240
|
+
# LALR/Earley disagreement is not confined to one exception class, so
|
|
241
|
+
# narrowing the retry to just UnexpectedToken would silently keep exactly the
|
|
242
|
+
# narrowing this wrapper exists to remove). Earley remains the ground truth of
|
|
243
|
+
# what the mode accepts, and what tree it builds, on every input; LALR only
|
|
244
|
+
# ever makes an accepted formula faster to parse, never changes what parses.
|
|
245
|
+
# See MSFLParser._parse_tree.
|
|
246
|
+
# ``tomodal`` inherits the modal mode's operators wholesale, so it inherits
|
|
247
|
+
# the reason too. ``modal_sorted`` (modal=True, many_sorted=True) ALSO
|
|
248
|
+
# inherits the modal mode's operators wholesale -- including the hybrid
|
|
249
|
+
# nominal rule -- via _clone_parser_ops_sorted (fol/nodes.py), so it inherits
|
|
250
|
+
# the same LALR/Earley conflict and needs the same fallback. ``to`` (classical
|
|
251
|
+
# third order) and ``so_sorted`` (second_order=True, many_sorted=True) do not
|
|
252
|
+
# inherit any hybrid/modal operator, and parse with plain LALR like the modes
|
|
253
|
+
# they extend.
|
|
254
|
+
_HYBRID_MODES = frozenset({"modal", "tomodal", "modal_sorted"})
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
# Modes carrying the agent-indexed epistemic/doxastic operators, whose free
|
|
258
|
+
# agent variables are resolved to named agents after the parse.
|
|
259
|
+
# ``resolve_agent_variables`` (fol/_modal_nodes.py) already tracks a bound
|
|
260
|
+
# object variable through SortedQuantifier as well as plain Quantifier, so
|
|
261
|
+
# ``modal_sorted`` needs no separate handling beyond being listed here.
|
|
262
|
+
_AGENT_MODES = frozenset({"modal", "tomodal", "modal_sorted"})
|
|
263
|
+
|
|
264
|
+
# The third-order modes, whose formulas are TYPE-CHECKED after the parse:
|
|
265
|
+
# an argument slot holds an individual or a property, never both, and only
|
|
266
|
+
# these modes can express the difference in the first place.
|
|
267
|
+
_THIRD_ORDER_MODES = frozenset({"to", "tomodal"})
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def _cached_parser(registry_mode: str, kind: str) -> Lark:
|
|
271
|
+
"""Return the compiled Lark parser for ``(registry_mode, kind)``, building
|
|
272
|
+
and caching it on first request. ``kind`` is ``"lalr"`` or ``"earley"``
|
|
273
|
+
(Lark's own ``parser=`` values) -- see ``_PARSER_CACHE`` and
|
|
274
|
+
``_HYBRID_MODES``. A hybrid mode calls this twice, once per kind, and gets
|
|
275
|
+
two independent cached parsers for the SAME grammar text.
|
|
276
|
+
"""
|
|
277
|
+
cache_key = (registry_mode, len(PARSER_OPS), kind)
|
|
278
|
+
parser = _PARSER_CACHE.get(cache_key)
|
|
279
|
+
if parser is None:
|
|
280
|
+
grammar_text = _allow_single_letter_function_calls(build_grammar(registry_mode))
|
|
281
|
+
# propagate_positions=True costs nothing parse()-observable — it only
|
|
282
|
+
# adds a `.meta` (start_pos/end_pos/line/column/end_line/end_column)
|
|
283
|
+
# to every Tree Lark builds, which plain parse() never looks at. It is
|
|
284
|
+
# what parse_with_spans (below) reads to build its SpanMap; verified
|
|
285
|
+
# empirically (see spans.py's module docstring / the "prove it on a
|
|
286
|
+
# real example" section of this change's report) that both Lark
|
|
287
|
+
# backends populate it accurately, including for Tokens, without
|
|
288
|
+
# needing anything beyond this flag.
|
|
289
|
+
parser = Lark(grammar_text, parser=kind,
|
|
290
|
+
import_paths=[str(_GRAMMARS_DIR)],
|
|
291
|
+
propagate_positions=True)
|
|
292
|
+
_PARSER_CACHE[cache_key] = parser
|
|
293
|
+
return parser
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def _assemble_transformer(registry_mode: str) -> LambdaTransformer:
|
|
298
|
+
"""Build the Transformer for a registry mode from the parser registry.
|
|
299
|
+
|
|
300
|
+
The base is a LambdaTransformer, which carries the shared, non-operator
|
|
301
|
+
term/atom/lambda/application handlers (VARIABLE, NAME, function_, atom_, sum,
|
|
302
|
+
product, lambda_, …) common to every mode. The mode's registered operator
|
|
303
|
+
handlers (build_transform_handlers) are then attached as INSTANCE attributes.
|
|
304
|
+
|
|
305
|
+
Instance — not class — attributes are essential: a plain ``transform(items)``
|
|
306
|
+
function attached to the instance is returned bare by ``getattr(self, alias)``,
|
|
307
|
+
so Lark calls it with the rule's children as the sole argument. Attached to the
|
|
308
|
+
class it would become a bound method and receive ``self`` as ``items``.
|
|
309
|
+
"""
|
|
310
|
+
transformer = LambdaTransformer()
|
|
311
|
+
for alias, fn in build_transform_handlers(registry_mode).items():
|
|
312
|
+
setattr(transformer, alias, fn)
|
|
313
|
+
return transformer
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
# =========================
|
|
317
|
+
# Span-capturing transform (used only by MSFLParser.parse_with_spans)
|
|
318
|
+
# =========================
|
|
319
|
+
#
|
|
320
|
+
# Lark's Transformer routes EVERY rule reduction through _call_userfunc(tree,
|
|
321
|
+
# new_children) and every token through _call_userfunc_token(token) (see
|
|
322
|
+
# lark.visitors.Transformer) — tree.meta (with propagate_positions=True, set
|
|
323
|
+
# above) and the token itself both already carry exact source positions.
|
|
324
|
+
# Overriding just these two dispatch hooks lets us record spans for whatever
|
|
325
|
+
# Node each already-existing handler produces, WITHOUT changing a single
|
|
326
|
+
# handler in FOLTransformer / LambdaTransformer / the operator registry: every
|
|
327
|
+
# rule alias still runs exactly the function build_transform_handlers attaches
|
|
328
|
+
# for plain parse() (level2/fold aliases are the one exception — see
|
|
329
|
+
# _fold_level2 below — and even those build the exact same node graph
|
|
330
|
+
# _fold_binary would, just with span bookkeeping alongside), we just
|
|
331
|
+
# additionally look at the Tree/Token that was being reduced once the
|
|
332
|
+
# (unmodified) handler has returned.
|
|
333
|
+
#
|
|
334
|
+
# TWO id()-keyed dicts are built (id_extent, id_head) — see spans.py's
|
|
335
|
+
# build_span_map docstring for why id() is fine here: it is a transient,
|
|
336
|
+
# same-call-only bookkeeping device, never exposed past this transform. Once
|
|
337
|
+
# the whole tree has been transformed, MSFLParser.parse_with_spans turns
|
|
338
|
+
# these into the PATH-keyed, spec-correct SpanMap it actually returns.
|
|
339
|
+
#
|
|
340
|
+
# HEAD SPANS — THE LARK-FILTERING PROBLEM AND WHY THIS DOES NOT USE
|
|
341
|
+
# keep_all_tokens=True
|
|
342
|
+
# ---------------------------------------------------------------------------
|
|
343
|
+
# Every classical connective (¬ ∧ ∨ → ↔ ⊕) and every infix comparison
|
|
344
|
+
# (= < > ≤ ≥ ≠) is declared as an ANONYMOUS string-literal terminal inline in
|
|
345
|
+
# the grammar (e.g. '"¬" prefix' -> not_), and Lark auto-filters an anonymous
|
|
346
|
+
# literal out of the parse tree by default — so its own token never reaches
|
|
347
|
+
# _call_userfunc's new_children, and there is no tree.meta for the head
|
|
348
|
+
# position on its own (only the whole rule's meta is available).
|
|
349
|
+
#
|
|
350
|
+
# The obvious fix, building the span parser with keep_all_tokens=True, was
|
|
351
|
+
# tried and rejected: it does not just ADD the filtered tokens back, it also
|
|
352
|
+
# defeats Lark's "?rule inlines away when it reduces to one child after
|
|
353
|
+
# filtering" mechanism that '?prefix: … | "(" formula ")"' relies on to make
|
|
354
|
+
# a parenthesised subexpression collapse straight through to its inner
|
|
355
|
+
# node — with keep_all_tokens=True the "(" and ")" survive as extra children,
|
|
356
|
+
# the alternative no longer has one child, so it stops inlining and instead
|
|
357
|
+
# becomes an un-aliased Tree('prefix', [...]) with NO transform handler
|
|
358
|
+
# (Transformer.__default__ returns it as a bare Tree, not a Node) — breaking
|
|
359
|
+
# AST construction for every parenthesised subformula. It was also going to
|
|
360
|
+
# require a SECOND compiled Lark parser (a second Earley grammar build),
|
|
361
|
+
# which the "do not change parse()'s performance" requirement rules out for
|
|
362
|
+
# the shared, cached self.parser — and a keep_all_tokens tree also does not
|
|
363
|
+
# feed the mode's OWN registered handlers (not_, and_, …) unmodified, since
|
|
364
|
+
# those handlers are written for the filtered item shape; reusing them would
|
|
365
|
+
# need re-deriving the filtering by hand anyway.
|
|
366
|
+
#
|
|
367
|
+
# So this uses the parser exactly as parse() does (propagate_positions,
|
|
368
|
+
# no keep_all_tokens, same cached Lark instance — parse()'s own behaviour and
|
|
369
|
+
# performance are untouched) and recovers each head position from what IS
|
|
370
|
+
# already available without keep_all_tokens:
|
|
371
|
+
# * FORALL/EXISTS, COUNTOP: NAMED terminals are never auto-filtered, so the
|
|
372
|
+
# raw Token is already sitting in new_children — no scan needed. A
|
|
373
|
+
# quantifier's head is [that token's start, the bound Variable's own
|
|
374
|
+
# already-known extent end) — which is exactly "symbol + variable +
|
|
375
|
+
# whatever whitespace sits between them", satisfying A5 directly.
|
|
376
|
+
# * PREDICATE (an atom's head): also a named terminal, same story — the raw
|
|
377
|
+
# token is new_children[0] for atom_/atom0_.
|
|
378
|
+
# * A prefix op's glyph (¬, and any future one driven by the OPERATORS
|
|
379
|
+
# registry's fixity=="prefix"): always the FIRST thing the rule matches
|
|
380
|
+
# (tree.meta.start_pos), so the head is simply
|
|
381
|
+
# [meta.start_pos, meta.start_pos + len(glyph)) — no scan needed either.
|
|
382
|
+
# * A binary infix glyph (→, ↔, and every comparison predicate) and a
|
|
383
|
+
# same-level fold's connective (∧, ∨, ⊕, Ⓒ): genuinely filtered, with no
|
|
384
|
+
# token anywhere to read a position off. These are located by
|
|
385
|
+
# spans.find_glyph, searching the SOURCE TEXT for the operator's own
|
|
386
|
+
# glyph within the gap between the two operands' own already-known
|
|
387
|
+
# (inner) extents. That gap can be WIDER than "just whitespace" — an
|
|
388
|
+
# operand may carry its own redundant wrapping parentheses, which its
|
|
389
|
+
# inner extent (per A5 item 3) deliberately excludes — but find_glyph
|
|
390
|
+
# searches for the EXACT glyph text rather than "the first non-whitespace
|
|
391
|
+
# run", which is what makes that safe (no parenthesis can equal the
|
|
392
|
+
# glyph). See spans.find_glyph's docstring.
|
|
393
|
+
#
|
|
394
|
+
# A SAME-LEVEL FOLD gets special handling on top of this (_fold_level2): only
|
|
395
|
+
# the OUTERMOST fold result is directly dispatched by Lark (see
|
|
396
|
+
# spans.py's module docstring for why), so every INTERMEDIATE left-fold node
|
|
397
|
+
# is built manually here, in lockstep with locating each connective
|
|
398
|
+
# occurrence — see _fold_level2's own docstring for why the intermediate's
|
|
399
|
+
# own EXTENT cannot be the naive union of its two operands' inner extents
|
|
400
|
+
# either (same "an operand may have its own wrapping parens" problem, this
|
|
401
|
+
# time for extent rather than head).
|
|
402
|
+
#
|
|
403
|
+
# This mixin sits BEFORE LambdaTransformer in MRO, so its super() calls reach
|
|
404
|
+
# Transformer's real dispatch, which looks up handlers via getattr(self, ...)
|
|
405
|
+
# — INSTANCE attributes (the setattr'd registry handlers) are found exactly as
|
|
406
|
+
# for a plain LambdaTransformer instance; only the dispatch hooks below differ.
|
|
407
|
+
|
|
408
|
+
#: rule_alias -> the exact glyph text for every "binary, single occurrence,
|
|
409
|
+
#: filtered-anonymous-literal" operator this transform locates via
|
|
410
|
+
#: spans.find_glyph: the classical infix comparisons and the arithmetic term
|
|
411
|
+
#: operators (neither driven by the OPERATORS registry, which only covers
|
|
412
|
+
#: FORMULA operators, not the term layer's infix predicates/functions) plus
|
|
413
|
+
#: the two binary connective levels (→, ↔ — these ARE registry-driven; see
|
|
414
|
+
#: _binary_glyph below, which merges this with the registry so both paths
|
|
415
|
+
#: share one lookup). The arithmetic entries give a Function("+"/"-"/"*"/"/",
|
|
416
|
+
#: …) node — built by the SAME shared term layer as any other function, so
|
|
417
|
+
#: its head is that operator occurrence, exactly like an infix comparison's.
|
|
418
|
+
_INFIX_TERM_GLYPH = {
|
|
419
|
+
"eq_": "=", "lt_": "<", "gt_": ">", "le_": "≤", "ge_": "≥", "ne_": "≠",
|
|
420
|
+
"add_": "+", "sub_": "-", "mul_": "*", "div_": "/",
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
class _SpanCapturingTransform:
|
|
425
|
+
"""Mixin: record precise (extent, head) spans for every Node a rule/token
|
|
426
|
+
handler builds, keyed transiently by id() — see the module comment above
|
|
427
|
+
and spans.py's ``build_span_map`` docstring for why id() here is safe.
|
|
428
|
+
|
|
429
|
+
``self.id_extent`` / ``self.id_head`` (``id(node) -> spans.Span``) are
|
|
430
|
+
populated as a side effect of ``transform()``; the caller
|
|
431
|
+
(``MSFLParser.parse_with_spans``) turns them into a path-keyed
|
|
432
|
+
:class:`~unicode_logic_kit.fol.spans.SpanMap` once transform() returns. See
|
|
433
|
+
spans.py's module docstring for the two cases this alone cannot recover
|
|
434
|
+
(an agent variable sliced out of a combined K_a-style token; a
|
|
435
|
+
higher-order lambda rewrite) — both report :data:`~unicode_logic_kit.fol.spans.UNKNOWN`.
|
|
436
|
+
"""
|
|
437
|
+
|
|
438
|
+
def __init__(self, text: str, registry_mode: str, *args, **kwargs):
|
|
439
|
+
super().__init__(*args, **kwargs)
|
|
440
|
+
self._span_text = text
|
|
441
|
+
self.id_extent: dict = {}
|
|
442
|
+
self.id_head: dict = {}
|
|
443
|
+
|
|
444
|
+
# Registry-driven lookup tables, built once per transformer instance
|
|
445
|
+
# from this mode's ParserOps: which rule aliases are same-level folds
|
|
446
|
+
# (and which Node class/glyph each folds into), and which are a
|
|
447
|
+
# non-folded connective driven by the OPERATORS registry (fixity ==
|
|
448
|
+
# "prefix" | "binary_implies" | "binary_iff" | "binary_until" — every
|
|
449
|
+
# OTHER fixity, notably "agent_prefix", is deliberately excluded: its
|
|
450
|
+
# token is not a bare glyph, see the module comment's UNKNOWN case).
|
|
451
|
+
self._level2_cls: dict = {}
|
|
452
|
+
self._binary_glyph: dict = dict(_INFIX_TERM_GLYPH)
|
|
453
|
+
for op in parser_ops_for_mode(registry_mode):
|
|
454
|
+
if op.level == "level2":
|
|
455
|
+
self._level2_cls[op.rule_alias] = op.node_class
|
|
456
|
+
continue
|
|
457
|
+
spec = OPERATORS.get(op.node_class.__name__)
|
|
458
|
+
if spec is not None and spec.fixity in (
|
|
459
|
+
"binary_implies", "binary_iff", "binary_until"):
|
|
460
|
+
self._binary_glyph[op.rule_alias] = spec.unicode
|
|
461
|
+
|
|
462
|
+
def _call_userfunc(self, tree, new_children=None):
|
|
463
|
+
text = self._span_text
|
|
464
|
+
alias = str(tree.data)
|
|
465
|
+
children = tree.children if new_children is None else new_children
|
|
466
|
+
|
|
467
|
+
if alias in self._level2_cls:
|
|
468
|
+
return self._fold_level2(alias, self._level2_cls[alias], children, tree.meta)
|
|
469
|
+
|
|
470
|
+
result = super()._call_userfunc(tree, children)
|
|
471
|
+
if not isinstance(result, Node):
|
|
472
|
+
return result
|
|
473
|
+
if not tree.meta.empty:
|
|
474
|
+
self.id_extent[id(result)] = span_from_meta(tree.meta, text)
|
|
475
|
+
self._assign_head(alias, tree, children, result)
|
|
476
|
+
return result
|
|
477
|
+
|
|
478
|
+
def _call_userfunc_token(self, token):
|
|
479
|
+
result = super()._call_userfunc_token(token)
|
|
480
|
+
if isinstance(result, Node):
|
|
481
|
+
span = span_from_token(token, self._span_text)
|
|
482
|
+
self.id_extent[id(result)] = span
|
|
483
|
+
self.id_head[id(result)] = span # a leaf term IS its own head
|
|
484
|
+
return result
|
|
485
|
+
|
|
486
|
+
# -- head-span derivation, per rule shape ----------------------------
|
|
487
|
+
|
|
488
|
+
def _assign_head(self, alias, tree, children, result):
|
|
489
|
+
"""Set ``self.id_head[id(result)]`` for one directly-dispatched
|
|
490
|
+
(non-fold) rule reduction, using whichever mechanism the module
|
|
491
|
+
comment above documents for ``alias``'s shape. A rule this covers
|
|
492
|
+
no case for (out of this change's target fragment) is simply left
|
|
493
|
+
without a head entry — reported as UNKNOWN, never guessed."""
|
|
494
|
+
text = self._span_text
|
|
495
|
+
|
|
496
|
+
if alias in ("atom_", "atom0_"):
|
|
497
|
+
# PREDICATE is a named terminal: never filtered, already a raw
|
|
498
|
+
# Token in children[0] (no per-token handler turns it into a Node).
|
|
499
|
+
tok = children[0]
|
|
500
|
+
if isinstance(tok, Token):
|
|
501
|
+
self.id_head[id(result)] = span_from_token(tok, text)
|
|
502
|
+
return
|
|
503
|
+
|
|
504
|
+
if alias in ("true_", "false_"):
|
|
505
|
+
# The glyph ``⊤`` / ``⊥`` is the whole atom, so its head is its extent.
|
|
506
|
+
extent = self.id_extent.get(id(result))
|
|
507
|
+
if extent is not None:
|
|
508
|
+
self.id_head[id(result)] = extent
|
|
509
|
+
return
|
|
510
|
+
|
|
511
|
+
if alias in ("const_", "number_"):
|
|
512
|
+
# CONSTANT / NUMBER also have no per-token handler, so the raw
|
|
513
|
+
# token — the leaf's ENTIRE span — is still sitting in children[0].
|
|
514
|
+
tok = children[0]
|
|
515
|
+
if isinstance(tok, Token):
|
|
516
|
+
span = span_from_token(tok, text)
|
|
517
|
+
self.id_extent[id(result)] = span
|
|
518
|
+
self.id_head[id(result)] = span
|
|
519
|
+
return
|
|
520
|
+
|
|
521
|
+
if alias == "function_":
|
|
522
|
+
# The function name occupies whatever leaf (Constant/Variable
|
|
523
|
+
# Node, already span-captured; or a raw Token in an edge case)
|
|
524
|
+
# was consumed as children[0] — that leaf's own span IS the
|
|
525
|
+
# function's head.
|
|
526
|
+
head_src = children[0]
|
|
527
|
+
if isinstance(head_src, Node) and id(head_src) in self.id_extent:
|
|
528
|
+
self.id_head[id(result)] = self.id_extent[id(head_src)]
|
|
529
|
+
elif isinstance(head_src, Token):
|
|
530
|
+
self.id_head[id(result)] = span_from_token(head_src, text)
|
|
531
|
+
return
|
|
532
|
+
|
|
533
|
+
if alias == "quantifier_":
|
|
534
|
+
self._assign_binder_head(children[0], children[1], result)
|
|
535
|
+
return
|
|
536
|
+
|
|
537
|
+
if alias == "count_":
|
|
538
|
+
# COUNTOP NUMBER VARIABLE prefix -> the head runs from COUNTOP's
|
|
539
|
+
# own start through the bound VARIABLE's end (COUNTOP and the
|
|
540
|
+
# bound NUMBER are both named terminals, but only the VARIABLE's
|
|
541
|
+
# end matters for the head's right edge — "∃≥3 y").
|
|
542
|
+
self._assign_binder_head(children[0], children[2], result)
|
|
543
|
+
return
|
|
544
|
+
|
|
545
|
+
if alias in self._binary_glyph:
|
|
546
|
+
left, right = children[0], children[1]
|
|
547
|
+
le = self.id_extent.get(id(left))
|
|
548
|
+
re_ = self.id_extent.get(id(right))
|
|
549
|
+
if le is not None and re_ is not None:
|
|
550
|
+
found = _safe_find_glyph(text, le.end, re_.start, self._binary_glyph[alias])
|
|
551
|
+
if found is not None:
|
|
552
|
+
self.id_head[id(result)] = make_span(text, *found)
|
|
553
|
+
return
|
|
554
|
+
|
|
555
|
+
spec = OPERATORS.get(type(result).__name__)
|
|
556
|
+
if spec is not None and spec.fixity == "prefix" and not tree.meta.empty:
|
|
557
|
+
s = tree.meta.start_pos
|
|
558
|
+
self.id_head[id(result)] = make_span(text, s, s + len(spec.unicode))
|
|
559
|
+
return
|
|
560
|
+
# Any other alias (agent_prefix, binders outside quantifier_/count_,
|
|
561
|
+
# lambda_/application_, out-of-fragment term forms, …): no rule here,
|
|
562
|
+
# head stays UNKNOWN for this node.
|
|
563
|
+
|
|
564
|
+
def _assign_binder_head(self, symbol_tok, var_node, result):
|
|
565
|
+
"""Shared by quantifier_/count_: head = [symbol_tok.start_pos,
|
|
566
|
+
var_node's own extent end) — the binder symbol together with its
|
|
567
|
+
bound variable, whitespace between them included (A5)."""
|
|
568
|
+
text = self._span_text
|
|
569
|
+
ve = self.id_extent.get(id(var_node))
|
|
570
|
+
if isinstance(symbol_tok, Token) and ve is not None:
|
|
571
|
+
self.id_head[id(result)] = make_span(text, symbol_tok.start_pos, ve.end)
|
|
572
|
+
|
|
573
|
+
# -- same-level operator-chain folding, WITH span bookkeeping ---------
|
|
574
|
+
|
|
575
|
+
def _fold_level2(self, alias, node_cls, items, meta):
|
|
576
|
+
"""Left-fold ``items`` into nested ``node_cls`` binary nodes — same
|
|
577
|
+
node graph ``FOLTransformer._fold_binary`` builds (deterministic,
|
|
578
|
+
pure function of ``items``/``node_cls``, so building it here instead
|
|
579
|
+
of delegating produces a structurally identical AST) — while
|
|
580
|
+
additionally recording (extent, head) for every INTERMEDIATE node,
|
|
581
|
+
not just the outermost one Lark itself dispatches (see the module
|
|
582
|
+
comment's "A SAME-LEVEL FOLD" section).
|
|
583
|
+
|
|
584
|
+
Each connective's HEAD is found once per gap, from the ORIGINAL
|
|
585
|
+
operands' own (already-known) inner extents — safe even when an
|
|
586
|
+
operand carries its own redundant wrapping parens, same reasoning as
|
|
587
|
+
find_glyph's docstring.
|
|
588
|
+
|
|
589
|
+
Each intermediate's EXTENT is NOT the naive union of its two
|
|
590
|
+
operands' inner extents — an operand's inner extent deliberately
|
|
591
|
+
excludes redundant wrapping parens (A5 item 3), so if operand k is
|
|
592
|
+
individually parenthesised, "union of inner extents" would slice
|
|
593
|
+
THROUGH that closing paren and the next operand's opening paren,
|
|
594
|
+
producing a non-reparseable fragment (e.g. "(P) ∧ (Q)" would wrongly
|
|
595
|
+
shrink to "P) ∧ (Q"). Instead every intermediate STARTS at the whole
|
|
596
|
+
fold's own start (meta.start_pos — exactly operand 0's true outer
|
|
597
|
+
bound, parens included, since operand 0 begins the rule's own match)
|
|
598
|
+
and ENDS either at the position right before the NEXT connective
|
|
599
|
+
(whitespace-trimmed backward — exactly operand k's true outer bound)
|
|
600
|
+
for a non-final intermediate, or at the whole fold's own end
|
|
601
|
+
(meta.end_pos) for the final one — both of which correctly include
|
|
602
|
+
any wrapping parens an operand happens to have, because they are
|
|
603
|
+
read off the ACTUAL SOURCE TEXT / the whole rule's own Lark-computed
|
|
604
|
+
bounds rather than reconstructed from the operands' narrowed inner
|
|
605
|
+
extents.
|
|
606
|
+
"""
|
|
607
|
+
text = self._span_text
|
|
608
|
+
glyph = OPERATORS[node_cls.__name__].unicode
|
|
609
|
+
n = len(items)
|
|
610
|
+
|
|
611
|
+
# Every connective's (start, end), found once, in left-to-right order.
|
|
612
|
+
glyph_spans = []
|
|
613
|
+
for j in range(n - 1):
|
|
614
|
+
le = self.id_extent.get(id(items[j]))
|
|
615
|
+
re_ = self.id_extent.get(id(items[j + 1]))
|
|
616
|
+
glyph_spans.append(
|
|
617
|
+
_safe_find_glyph(text, le.end, re_.start, glyph)
|
|
618
|
+
if le is not None and re_ is not None else None)
|
|
619
|
+
|
|
620
|
+
outer_start = meta.start_pos if not meta.empty else None
|
|
621
|
+
outer_end = meta.end_pos if not meta.empty else None
|
|
622
|
+
|
|
623
|
+
node = items[0]
|
|
624
|
+
for j in range(n - 1):
|
|
625
|
+
node = node_cls(node, items[j + 1])
|
|
626
|
+
gspan = glyph_spans[j]
|
|
627
|
+
if gspan is not None:
|
|
628
|
+
self.id_head[id(node)] = make_span(text, *gspan)
|
|
629
|
+
if outer_start is None:
|
|
630
|
+
continue # no rule meta at all (should not happen) -> extent left to fill_gap_spans
|
|
631
|
+
if j == n - 2:
|
|
632
|
+
end = outer_end
|
|
633
|
+
elif glyph_spans[j + 1] is not None:
|
|
634
|
+
end = trim_ws_backward(text, glyph_spans[j + 1][0])
|
|
635
|
+
else:
|
|
636
|
+
end = None
|
|
637
|
+
if end is not None:
|
|
638
|
+
self.id_extent[id(node)] = make_span(text, outer_start, end)
|
|
639
|
+
return node
|
|
640
|
+
|
|
641
|
+
|
|
642
|
+
class _SpanLambdaTransformer(_SpanCapturingTransform, LambdaTransformer):
|
|
643
|
+
"""LambdaTransformer with span capture mixed in; see _SpanCapturingTransform."""
|
|
644
|
+
|
|
645
|
+
def lambda_(self, items):
|
|
646
|
+
"""``LambdaTransformer.lambda_`` plus a span for the bound parameter.
|
|
647
|
+
|
|
648
|
+
The base handler reads the parameter out of ``items[1]`` and then
|
|
649
|
+
builds a FRESH ``LambdaVar`` from its name, discarding the node (or
|
|
650
|
+
raw PREDICATE token) it came from. That new object never passes
|
|
651
|
+
through Lark's dispatch hooks, so the mixin never sees it — and
|
|
652
|
+
because ``LambdaVar`` is a leaf, ``fill_gap_spans`` cannot derive
|
|
653
|
+
its span from children either. Without this override the parameter
|
|
654
|
+
of EVERY lambda, ordinary ones included, would report UNKNOWN while
|
|
655
|
+
its source position (``λx. …`` -> the ``x``) sits right there in
|
|
656
|
+
``items[1]``.
|
|
657
|
+
|
|
658
|
+
Overridden here rather than in ``LambdaTransformer`` on purpose:
|
|
659
|
+
this subclass exists only for the span pass, so plain ``parse()``
|
|
660
|
+
keeps running the untouched handler.
|
|
661
|
+
"""
|
|
662
|
+
result = super().lambda_(items)
|
|
663
|
+
param_source = items[1]
|
|
664
|
+
span = None
|
|
665
|
+
if isinstance(param_source, Node):
|
|
666
|
+
span = self.id_extent.get(id(param_source))
|
|
667
|
+
elif isinstance(param_source, Token):
|
|
668
|
+
span = span_from_token(param_source, self._span_text)
|
|
669
|
+
if span is not None and isinstance(result, Lambda):
|
|
670
|
+
self.id_extent[id(result.param)] = span
|
|
671
|
+
self.id_head[id(result.param)] = span # leaf: head == extent
|
|
672
|
+
return result
|
|
673
|
+
|
|
674
|
+
|
|
675
|
+
def _assemble_span_transformer(registry_mode: str, text: str) -> _SpanLambdaTransformer:
|
|
676
|
+
"""Span-capturing sibling of _assemble_transformer, for the same registry_mode.
|
|
677
|
+
|
|
678
|
+
Attaches the SAME per-mode operator handlers (build_transform_handlers) as
|
|
679
|
+
_assemble_transformer — parse_with_spans builds an AST byte-identical to
|
|
680
|
+
parse()'s, just with spans recorded alongside.
|
|
681
|
+
"""
|
|
682
|
+
transformer = _SpanLambdaTransformer(text, registry_mode)
|
|
683
|
+
for alias, fn in build_transform_handlers(registry_mode).items():
|
|
684
|
+
setattr(transformer, alias, fn)
|
|
685
|
+
return transformer
|
|
686
|
+
|
|
687
|
+
|
|
688
|
+
class MSFLParser:
|
|
689
|
+
"""Unified parser supporting FOL, MSFOL, MSFL, FL, modal, and second-order modes.
|
|
690
|
+
|
|
691
|
+
Args:
|
|
692
|
+
many_sorted: if True, quantifiers and constants must carry sort
|
|
693
|
+
annotations (e.g. ``∀x:Human P(x)``, ``alice:Human``).
|
|
694
|
+
fuzzy: if True, use Łukasiewicz operators (⊗ ⊕ for strong
|
|
695
|
+
conjunction/disjunction; ¬ → ↔ map to Łukasiewicz nodes).
|
|
696
|
+
modal: if True, parse FOL extended with modal, epistemic, doxastic,
|
|
697
|
+
temporal, and deontic operators (□ ◇ K_a B_a Ⓖ Ⓕ Ⓝ Ⓤ Ⓞ Ⓟ) over
|
|
698
|
+
classical unsorted quantifiers/constants, or — combined with
|
|
699
|
+
many_sorted=True — over SORTED quantifiers/constants (every
|
|
700
|
+
binder then requires a sort annotation, exactly like plain
|
|
701
|
+
MSFOL: ``□∀x:Human (Mortal(x))``). Cannot be combined with fuzzy
|
|
702
|
+
in v1.
|
|
703
|
+
second_order: if True, parse classical FOL extended with
|
|
704
|
+
second-order quantifiers over predicate variables (∀P / ∃P, where P
|
|
705
|
+
is an uppercase PREDICATE; the bound predicate's arity is inferred
|
|
706
|
+
from its applications in the body) over classical unsorted
|
|
707
|
+
individual quantifiers/constants, or — combined with
|
|
708
|
+
many_sorted=True — over SORTED ones (the predicate quantifier
|
|
709
|
+
itself stays unsorted; only the ∀x/∃x individual binders and bare
|
|
710
|
+
constants need a sort). Cannot be combined with fuzzy or modal
|
|
711
|
+
in v1.
|
|
712
|
+
third_order: if True, parse second-order syntax extended with predicates
|
|
713
|
+
in ARGUMENT position — ``Positive(G)``, ``Essence(G, x)``,
|
|
714
|
+
``Positive(λx. ¬G(x))``, which parse to an Atom over a
|
|
715
|
+
:class:`~unicode_logic_kit.fol.nodes.PredicateTerm` or a Lambda. This
|
|
716
|
+
is the level second-order quantification cannot reach, since it is a
|
|
717
|
+
change to the argument layer rather than another binder. Combines
|
|
718
|
+
with ``modal=True`` (third-order modal logic — the setting Gödel's
|
|
719
|
+
ontological argument is stated in); not with ``second_order`` (which
|
|
720
|
+
it contains), ``many_sorted`` or ``fuzzy``.
|
|
721
|
+
dependence: if True, parse the team-semantic dependence/IF fragment —
|
|
722
|
+
literals, ∧, splitting ∨, ∀/∃, dependence atoms ``=(x, y)``, and
|
|
723
|
+
slashed existentials ``∃y/{x} φ``. Standalone (no other flag).
|
|
724
|
+
linear: if True, parse propositional intuitionistic linear logic —
|
|
725
|
+
``⊗ & ⊕ ⊸ ! 𝟙`` over atomic propositions. Standalone.
|
|
726
|
+
lambek: if True, parse Lambek-calculus category types — ``• \\ /`` over
|
|
727
|
+
atomic categories (``NP``, ``S``, …). Standalone.
|
|
728
|
+
|
|
729
|
+
Mode matrix:
|
|
730
|
+
(False, False) → FOL: classical ops incl. xor (⊕), unsorted quantifiers/constants
|
|
731
|
+
(True, False) → MSFOL: classical ∧∨¬→↔⊕, sorted quantifiers/constants
|
|
732
|
+
(True, True) → MSFL: Łukasiewicz operators, sorted quantifiers/constants
|
|
733
|
+
(False, True) → FL: Łukasiewicz operators, unsorted quantifiers/constants
|
|
734
|
+
modal=True → MODAL: classical unsorted FOL + modal/temporal/hybrid operators
|
|
735
|
+
modal+many_sorted → MSMODAL: MODAL over SORTED quantifiers/constants
|
|
736
|
+
second_order=True → SO: classical unsorted FOL + second-order quantifiers (∀P / ∃P)
|
|
737
|
+
second_order+many_sorted → MSSO: SO over SORTED individual quantifiers/constants
|
|
738
|
+
third_order=True → TO: SO + predicates in argument position (Positive(G))
|
|
739
|
+
third_order+modal → TOM: TO + the modal operator family
|
|
740
|
+
dependence=True → DEP: team-semantic dependence/IF fragment
|
|
741
|
+
linear=True → ILL: propositional intuitionistic linear logic
|
|
742
|
+
lambek=True → L: Lambek-calculus category types
|
|
743
|
+
|
|
744
|
+
``modal+many_sorted`` and ``second_order+many_sorted`` are NOT third-order:
|
|
745
|
+
combining many_sorted with third_order stays refused (how a sort interacts
|
|
746
|
+
with third-order's individual-vs-property "slot" inference is a separate,
|
|
747
|
+
open design question) — use third_order (optionally with modal=True) for
|
|
748
|
+
the third-order legs, and many_sorted with at most one of modal /
|
|
749
|
+
second_order for the sorted legs.
|
|
750
|
+
"""
|
|
751
|
+
|
|
752
|
+
def __init__(self, many_sorted: bool = False, fuzzy: bool = False,
|
|
753
|
+
modal: bool = False, second_order: bool = False,
|
|
754
|
+
third_order: bool = False,
|
|
755
|
+
dependence: bool = False, linear: bool = False,
|
|
756
|
+
lambek: bool = False):
|
|
757
|
+
_exclusive = [name for name, flag in (
|
|
758
|
+
("dependence", dependence), ("linear", linear), ("lambek", lambek),
|
|
759
|
+
) if flag]
|
|
760
|
+
if _exclusive and (many_sorted or fuzzy or modal or second_order
|
|
761
|
+
or third_order or len(_exclusive) > 1):
|
|
762
|
+
raise ValueError(
|
|
763
|
+
f"{_exclusive[0]}=True cannot be combined with any other mode "
|
|
764
|
+
"flag; the dependence / linear / lambek modes are standalone "
|
|
765
|
+
"logics with their own connectives and semantics."
|
|
766
|
+
)
|
|
767
|
+
if dependence:
|
|
768
|
+
self._mode = "dependence"
|
|
769
|
+
elif linear:
|
|
770
|
+
self._mode = "linear"
|
|
771
|
+
elif lambek:
|
|
772
|
+
self._mode = "lambek"
|
|
773
|
+
elif third_order:
|
|
774
|
+
if many_sorted or fuzzy or second_order:
|
|
775
|
+
raise ValueError(
|
|
776
|
+
"third_order=True cannot be combined with many_sorted, fuzzy, "
|
|
777
|
+
"or second_order; third-order mode already CONTAINS "
|
|
778
|
+
"second-order syntax (∀P / ∃P) and adds predicates in "
|
|
779
|
+
"argument position on top of it. Combine it with modal=True "
|
|
780
|
+
"for third-order modal logic."
|
|
781
|
+
)
|
|
782
|
+
self._mode = "tomodal" if modal else "to"
|
|
783
|
+
elif second_order:
|
|
784
|
+
if fuzzy or modal:
|
|
785
|
+
raise ValueError(
|
|
786
|
+
"second_order=True cannot be combined with fuzzy or modal in "
|
|
787
|
+
"v1; second-order mode is FOL plus second-order quantifiers "
|
|
788
|
+
"over predicate variables (optionally sorted, via "
|
|
789
|
+
"many_sorted=True, on the individual side). Use "
|
|
790
|
+
"third_order=True (optionally with modal=True) for the mode "
|
|
791
|
+
"that combines second-order syntax with modal operators."
|
|
792
|
+
)
|
|
793
|
+
self._mode = "so_sorted" if many_sorted else "so"
|
|
794
|
+
elif modal:
|
|
795
|
+
if fuzzy:
|
|
796
|
+
raise ValueError(
|
|
797
|
+
"modal=True cannot be combined with fuzzy in v1; modal mode "
|
|
798
|
+
"is FOL plus modal operators (optionally many-sorted, via "
|
|
799
|
+
"many_sorted=True)."
|
|
800
|
+
)
|
|
801
|
+
self._mode = "modal_sorted" if many_sorted else "modal"
|
|
802
|
+
elif not many_sorted and not fuzzy:
|
|
803
|
+
self._mode = "fol"
|
|
804
|
+
elif many_sorted and not fuzzy:
|
|
805
|
+
self._mode = "msfol"
|
|
806
|
+
elif many_sorted and fuzzy:
|
|
807
|
+
self._mode = "msfl"
|
|
808
|
+
else:
|
|
809
|
+
self._mode = "fl"
|
|
810
|
+
|
|
811
|
+
# The grammar string and the matching Transformer are both assembled from
|
|
812
|
+
# the operator registry for this mode — no per-mode .lark file or
|
|
813
|
+
# hand-written Transformer subclass. The relative ``%import .terminals`` in
|
|
814
|
+
# the generated grammar resolves against the grammars directory. The
|
|
815
|
+
# registry grammar is further patched to allow single-letter
|
|
816
|
+
# (VARIABLE-headed) function calls — see
|
|
817
|
+
# _allow_single_letter_function_calls above.
|
|
818
|
+
registry_mode = _REGISTRY_MODE[self._mode]
|
|
819
|
+
self._hybrid = self._mode in _HYBRID_MODES
|
|
820
|
+
# self.parser is public: NamingError/ParsingError use parser.terminals
|
|
821
|
+
# and parser.lex(). It is always the LALR parser now -- the fast path
|
|
822
|
+
# for every mode, hybrid or not (see _cached_parser / _HYBRID_MODES).
|
|
823
|
+
self.parser = _cached_parser(registry_mode, "lalr")
|
|
824
|
+
# The Earley fallback, built only for a hybrid mode; None otherwise.
|
|
825
|
+
# See _parse_tree for how the two are combined.
|
|
826
|
+
self._earley_parser = _cached_parser(registry_mode, "earley") if self._hybrid else None
|
|
827
|
+
self._transformer = _assemble_transformer(registry_mode)
|
|
828
|
+
self._registry_mode = registry_mode # parse_with_spans re-derives its own transformer from this
|
|
829
|
+
|
|
830
|
+
def _parse_tree(self, text: str):
|
|
831
|
+
"""Parse ``text`` into a raw Lark ``Tree`` with this mode's LALR
|
|
832
|
+
parser, falling back to the Earley parser for a hybrid mode
|
|
833
|
+
(``_HYBRID_MODES``) when LALR refuses.
|
|
834
|
+
|
|
835
|
+
For a hybrid mode, ALL THREE of Lark's failure exceptions
|
|
836
|
+
(``UnexpectedCharacters``/``UnexpectedToken``/``UnexpectedEOF``)
|
|
837
|
+
trigger the retry -- the LALR/Earley disagreement documented at
|
|
838
|
+
``_HYBRID_MODES`` is not confined to one exception class, so
|
|
839
|
+
narrowing the retry to just ``UnexpectedToken`` would silently keep
|
|
840
|
+
exactly the narrowing this wrapper exists to remove.
|
|
841
|
+
|
|
842
|
+
If Earley ALSO fails, its exception is tagged
|
|
843
|
+
(``_from_earley_fallback = True``) before it propagates, so
|
|
844
|
+
``parse``/``parse_with_spans`` (via ``_error_parser``/
|
|
845
|
+
``_token_failure``) know to report it untranslated -- through the
|
|
846
|
+
Earley parser and without the LALR-shape reinterpretation below --
|
|
847
|
+
exactly as this mode's error model behaved before this wrapper
|
|
848
|
+
existed, since that mapping was calibrated for LALR failures only.
|
|
849
|
+
"""
|
|
850
|
+
if not self._hybrid:
|
|
851
|
+
return self.parser.parse(text)
|
|
852
|
+
try:
|
|
853
|
+
return self.parser.parse(text)
|
|
854
|
+
except (UnexpectedCharacters, UnexpectedToken, UnexpectedEOF):
|
|
855
|
+
try:
|
|
856
|
+
return self._earley_parser.parse(text)
|
|
857
|
+
except (UnexpectedCharacters, UnexpectedToken, UnexpectedEOF) as exc:
|
|
858
|
+
exc._from_earley_fallback = True
|
|
859
|
+
raise
|
|
860
|
+
|
|
861
|
+
def _error_parser(self, exc) -> Lark:
|
|
862
|
+
"""Which parser instance produced ``exc``: the Earley fallback if
|
|
863
|
+
``_parse_tree`` tagged it, else this mode's own (LALR) parser. Used to
|
|
864
|
+
build NamingError/ParsingError against the terminal patterns of the
|
|
865
|
+
parser that actually raised, not always the primary one."""
|
|
866
|
+
if getattr(exc, "_from_earley_fallback", False):
|
|
867
|
+
return self._earley_parser
|
|
868
|
+
return self.parser
|
|
869
|
+
|
|
870
|
+
def _token_failure(self, exc: UnexpectedToken, text: str):
|
|
871
|
+
"""Turn lark's "cannot shift this token" into the kit's error model.
|
|
872
|
+
|
|
873
|
+
The model MSFLParser publishes is a two-way split: NamingError for a
|
|
874
|
+
lexer-level failure, ParsingError for a structurally incomplete
|
|
875
|
+
formula. Under Earley's dynamic lexer that split fell out of lark's own
|
|
876
|
+
exception types, because the lexer only ever offers tokens the parser
|
|
877
|
+
can currently use -- a well-formed symbol in the wrong place never
|
|
878
|
+
became a token at all, it stayed an unscannable character. A table
|
|
879
|
+
lexer has no such scruples: it tokenises first and the parser refuses
|
|
880
|
+
afterwards, so the SAME input arrives here as UnexpectedToken.
|
|
881
|
+
|
|
882
|
+
For a hybrid mode whose Earley FALLBACK is the one that raised this
|
|
883
|
+
(``_from_earley_fallback``), no translation applies at all: that
|
|
884
|
+
exception already carries Earley's own semantics (see
|
|
885
|
+
``_parse_tree``), so it is reported exactly as the old, pre-wrapper,
|
|
886
|
+
pure-Earley modal parser would have.
|
|
887
|
+
|
|
888
|
+
The translation below is not a guess. Measured over the 1310-line FOLIO
|
|
889
|
+
fixture plus hand-written malformed shapes, the two parsers' failures
|
|
890
|
+
line up exactly, with no overlap in either direction:
|
|
891
|
+
|
|
892
|
+
Earley UnexpectedCharacters -> LALR UnexpectedToken, token != $END
|
|
893
|
+
Earley UnexpectedEOF -> LALR UnexpectedToken, token == $END
|
|
894
|
+
|
|
895
|
+
So $END means "the formula ended too early" (ParsingError) and anything
|
|
896
|
+
else means "this does not belong here" (NamingError), reported against
|
|
897
|
+
the offending token's FIRST CHARACTER and position -- which is the
|
|
898
|
+
character Earley used to name, on a real lark.UnexpectedCharacters so
|
|
899
|
+
that the UnexpectedInput API NamingError inherits (match_examples()
|
|
900
|
+
above all) keeps working. Of the 58 inputs rejected in fol mode, 26
|
|
901
|
+
messages come out byte-identical; the 32 that differ are all
|
|
902
|
+
"Incomplete formula ...
|
|
903
|
+
Expected: ...", where LALR knows the answer more precisely and stops
|
|
904
|
+
leaking an internal ``__ANON_5`` terminal name into user-facing text.
|
|
905
|
+
|
|
906
|
+
Earley parsers keep the old routing: their UnexpectedToken means
|
|
907
|
+
something else, and this mapping is calibrated for LALR only.
|
|
908
|
+
"""
|
|
909
|
+
if getattr(exc, "_from_earley_fallback", False):
|
|
910
|
+
return ParsingError(self._earley_parser, exc, text, mode=self._mode)
|
|
911
|
+
if exc.token.type == "$END":
|
|
912
|
+
# Re-shaped as an end-of-input failure so ParsingError takes its
|
|
913
|
+
# "Incomplete formula" branch rather than reporting an unexpected
|
|
914
|
+
# token whose text is the empty string.
|
|
915
|
+
return ParsingError(self.parser, UnexpectedEOF(sorted(exc.expected)),
|
|
916
|
+
text, mode=self._mode)
|
|
917
|
+
# A REAL lark.UnexpectedCharacters, not a stand-in with the four
|
|
918
|
+
# attributes NamingError happens to read. NamingError subclasses
|
|
919
|
+
# UnexpectedCharacters and copies the original's __dict__ onto itself,
|
|
920
|
+
# so callers inherit lark's UnexpectedInput API -- `match_examples()`
|
|
921
|
+
# above all, which classifies an error by re-parsing labelled examples
|
|
922
|
+
# and needs `state`, `considered_rules` and the rest. An audit caught
|
|
923
|
+
# the stand-in version of this: it left those attributes absent, so
|
|
924
|
+
# `err.match_examples(...)` raised AttributeError in the eight LALR
|
|
925
|
+
# modes while still working in modal. Building the real thing costs one
|
|
926
|
+
# slice of `text` and removes the whole class of question.
|
|
927
|
+
return NamingError(
|
|
928
|
+
self.parser,
|
|
929
|
+
UnexpectedCharacters(
|
|
930
|
+
text, exc.token.start_pos, exc.token.line, exc.token.column,
|
|
931
|
+
allowed=getattr(exc, "accepts", None) or exc.expected,
|
|
932
|
+
considered_tokens=None,
|
|
933
|
+
state=exc.state,
|
|
934
|
+
token_history=[exc.token],
|
|
935
|
+
terminals_by_name=self.parser.lexer_conf.terminals_by_name,
|
|
936
|
+
considered_rules=None,
|
|
937
|
+
),
|
|
938
|
+
text, mode=self._mode)
|
|
939
|
+
|
|
940
|
+
def parse(self, text: str) -> Node:
|
|
941
|
+
"""Parse a formula string and return an AST node.
|
|
942
|
+
|
|
943
|
+
Raises:
|
|
944
|
+
NamingError: lexer-level failure (unrecognized character).
|
|
945
|
+
ParsingError: parser-level failure (unexpected token or EOF), or a
|
|
946
|
+
transformation-level failure such as a second-order predicate
|
|
947
|
+
variable applied at conflicting arities (ConflictingArityError).
|
|
948
|
+
"""
|
|
949
|
+
try:
|
|
950
|
+
tree = self._parse_tree(text)
|
|
951
|
+
ast = self._transformer.transform(tree)
|
|
952
|
+
ast = resolve_lambda_scope(ast)
|
|
953
|
+
if self._mode in _AGENT_MODES:
|
|
954
|
+
# A free epistemic/doxastic agent variable (K_a) denotes a named agent;
|
|
955
|
+
# only an agent bound by an enclosing quantifier stays a variable.
|
|
956
|
+
ast = resolve_agent_variables(ast)
|
|
957
|
+
if self._mode in _THIRD_ORDER_MODES:
|
|
958
|
+
# Well-typedness of the argument layer is a parse-time
|
|
959
|
+
# question here: the grammar admits both an individual and
|
|
960
|
+
# a property in the same slot, and only the signature
|
|
961
|
+
# analysis can see that one predicate got both.
|
|
962
|
+
analyse_signatures([ast])
|
|
963
|
+
return ast
|
|
964
|
+
except UnexpectedCharacters as e:
|
|
965
|
+
raise NamingError(self._error_parser(e), e, text, mode=self._mode)
|
|
966
|
+
except UnexpectedToken as e:
|
|
967
|
+
raise self._token_failure(e, text)
|
|
968
|
+
except UnexpectedEOF as e:
|
|
969
|
+
raise ParsingError(self._error_parser(e), e, text, mode=self._mode)
|
|
970
|
+
except VisitError as e:
|
|
971
|
+
# A transformer handler raised. Surface a ParsingError it produced
|
|
972
|
+
# (e.g. ConflictingArityError from second-order arity inference)
|
|
973
|
+
# directly, rather than the opaque Lark VisitError wrapper.
|
|
974
|
+
if isinstance(e.orig_exc, ParsingError):
|
|
975
|
+
raise e.orig_exc
|
|
976
|
+
raise
|
|
977
|
+
|
|
978
|
+
def parse_with_spans(self, text: str) -> SpannedFormula:
|
|
979
|
+
"""Parse ``text`` like :meth:`parse`, and also return a source-span side table.
|
|
980
|
+
|
|
981
|
+
Returns a :class:`~unicode_logic_kit.fol.spans.SpannedFormula` — ``.formula``
|
|
982
|
+
is exactly what ``parse(text)`` would return (same AST, unchanged; this
|
|
983
|
+
method runs the SAME ``self._parse_tree`` LALR-first/Earley-fallback
|
|
984
|
+
lookup (see ``_HYBRID_MODES``) and the same
|
|
985
|
+
``resolve_lambda_scope``/``resolve_agent_variables`` rewrites — the span
|
|
986
|
+
bookkeeping is additive, nothing about how the AST itself is built
|
|
987
|
+
changes), and ``.spans`` is a :class:`~unicode_logic_kit.fol.spans.SpanMap`
|
|
988
|
+
keyed by PATH — see :mod:`unicode_logic_kit.fol.spans`'s module docstring
|
|
989
|
+
for the path convention and why a path (not a node, not an id()) is the
|
|
990
|
+
key. Raises the same exceptions as :meth:`parse`, for the same reasons.
|
|
991
|
+
|
|
992
|
+
Look a node's spans up either via its path (``spans.get(path)``, ``path``
|
|
993
|
+
from :func:`~unicode_logic_kit.fol.spans.traverse`) or, for a node object
|
|
994
|
+
already in hand, ``spans.for_node(node)`` — see
|
|
995
|
+
:class:`~unicode_logic_kit.fol.spans.SpanMap`. Either field of the
|
|
996
|
+
returned :class:`~unicode_logic_kit.fol.spans.NodeSpans` (``.extent``/
|
|
997
|
+
``.head``) may individually be :data:`~unicode_logic_kit.fol.spans.UNKNOWN`
|
|
998
|
+
— never a guessed or interpolated span — for a documented, narrow set of
|
|
999
|
+
cases (out-of-fragment operators; an agent variable sliced out of a
|
|
1000
|
+
combined ``K_a``-style token; a higher-order lambda application
|
|
1001
|
+
rewritten into a fresh Application/LambdaVar chain the original parse
|
|
1002
|
+
never produced) — see spans.py's module docstring for the full
|
|
1003
|
+
case-by-case argument. For the classical FOL fragment (∀ ∃ ¬ ∧ ∨ → ↔ ⊕,
|
|
1004
|
+
predicates over constants/variables/function terms) both fields are
|
|
1005
|
+
recovered exactly for every node.
|
|
1006
|
+
"""
|
|
1007
|
+
try:
|
|
1008
|
+
tree = self._parse_tree(text)
|
|
1009
|
+
span_transformer = _assemble_span_transformer(self._registry_mode, text)
|
|
1010
|
+
pre_ast = span_transformer.transform(tree)
|
|
1011
|
+
id_extent = span_transformer.id_extent
|
|
1012
|
+
id_head = span_transformer.id_head
|
|
1013
|
+
fill_gap_spans(pre_ast, id_extent, text)
|
|
1014
|
+
spans = build_span_map(pre_ast, id_extent, id_head)
|
|
1015
|
+
ast = resolve_lambda_scope(pre_ast)
|
|
1016
|
+
spans = project_spans(pre_ast, ast, spans)
|
|
1017
|
+
if self._mode in _AGENT_MODES:
|
|
1018
|
+
post_ast = resolve_agent_variables(ast)
|
|
1019
|
+
spans = project_spans(ast, post_ast, spans)
|
|
1020
|
+
ast = post_ast
|
|
1021
|
+
if self._mode in _THIRD_ORDER_MODES:
|
|
1022
|
+
analyse_signatures([ast])
|
|
1023
|
+
return SpannedFormula(ast, spans)
|
|
1024
|
+
except UnexpectedCharacters as e:
|
|
1025
|
+
raise NamingError(self._error_parser(e), e, text, mode=self._mode)
|
|
1026
|
+
except UnexpectedToken as e:
|
|
1027
|
+
raise self._token_failure(e, text)
|
|
1028
|
+
except UnexpectedEOF as e:
|
|
1029
|
+
raise ParsingError(self._error_parser(e), e, text, mode=self._mode)
|
|
1030
|
+
except VisitError as e:
|
|
1031
|
+
if isinstance(e.orig_exc, ParsingError):
|
|
1032
|
+
raise e.orig_exc
|
|
1033
|
+
raise
|