unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,422 @@
|
|
|
1
|
+
import weakref
|
|
2
|
+
from copy import copy
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
from lark import Lark, UnexpectedCharacters, UnexpectedToken, UnexpectedEOF
|
|
6
|
+
from lark.exceptions import ParseError
|
|
7
|
+
|
|
8
|
+
from ._identifiers import HUMAN_READABLE_PATTERNS
|
|
9
|
+
|
|
10
|
+
try:
|
|
11
|
+
from lark.lexer import BasicLexer as _BasicLexer, LexerThread as _LexerThread
|
|
12
|
+
except ImportError: # pragma: no cover - lark renamed its lexer internals
|
|
13
|
+
_BasicLexer = _LexerThread = None
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
# --- Re-lexing the prefix of a failed parse ---------------------------------
|
|
17
|
+
#
|
|
18
|
+
# NamingError names the token in FRONT of the offending character ("Invalid
|
|
19
|
+
# predicate 'Foo' - unexpected character ..."), so it has to tokenize the text
|
|
20
|
+
# up to the failure point. The exception lark raises cannot supply that: the
|
|
21
|
+
# Earley scanner leaves `token_history` at None, so all it knows is the
|
|
22
|
+
# character and the position.
|
|
23
|
+
#
|
|
24
|
+
# The obvious way to get the rest, `Lark.lex()`, is a trap for this grammar:
|
|
25
|
+
#
|
|
26
|
+
# * the modal dialect falls back to parser="earley", which lexes dynamically
|
|
27
|
+
# and keeps no standing lexer, so `Lark.lex` finds no `self.lexer` and
|
|
28
|
+
# constructs a fresh BasicLexer on EVERY call (the other dialects use LALR
|
|
29
|
+
# with a standing contextual lexer, but a message is built the same way
|
|
30
|
+
# for all of them);
|
|
31
|
+
# * BasicLexer.__init__ runs a terminal-collision check whenever the package
|
|
32
|
+
# `interegular` merely happens to be importable (lark/lexer.py, `if
|
|
33
|
+
# has_interegular:`), comparing every pair of same-priority terminal
|
|
34
|
+
# regexes;
|
|
35
|
+
# * PREDICATE/NAME/CONSTANT/VARIABLE/SORT are generated at import from the
|
|
36
|
+
# running interpreter's Unicode tables (see _identifiers.py), and comparing
|
|
37
|
+
# THOSE takes MINUTES: measured at 185 s for one failed parse where the
|
|
38
|
+
# same call takes 13 ms with interegular absent.
|
|
39
|
+
#
|
|
40
|
+
# Note what the third point does NOT say. The generated patterns are not
|
|
41
|
+
# enormous as text -- measured across all nine grammars the kit builds, the
|
|
42
|
+
# largest is NAME at 4856 characters and the rest are 1.0 to 1.8 kB. The cost
|
|
43
|
+
# is not in their length. interegular decides collisions by building a finite
|
|
44
|
+
# automaton per pattern and intersecting them pairwise, and these patterns
|
|
45
|
+
# range over most of the Unicode letter repertoire, so the automata are wide
|
|
46
|
+
# even where the source text is short. Do not "explain" this by their size:
|
|
47
|
+
# an earlier draft of this comment did, and the number it quoted was left over
|
|
48
|
+
# from the pre-0.23.0 enumerated character classes, which really were 192 kB.
|
|
49
|
+
#
|
|
50
|
+
# Nobody opts into that. `interegular` arrives as a transitive dependency of
|
|
51
|
+
# vLLM (via outlines), so every environment that evaluates model-generated
|
|
52
|
+
# formulas has it without asking, and model output fails to parse routinely
|
|
53
|
+
# rather than exceptionally. The cost is also invisible: the call looks like an
|
|
54
|
+
# ordinary parse and takes four orders of magnitude longer, which reads as a
|
|
55
|
+
# hung worker, not as a slow error message.
|
|
56
|
+
#
|
|
57
|
+
# So the lexer used for messages is built ONCE per parser and kept, with
|
|
58
|
+
# validation off. Both halves earn their place: caching pays the check once per
|
|
59
|
+
# parser instead of once per failure, and skipping it removes the check
|
|
60
|
+
# altogether. Nothing is lost by skipping. Validation checks that the GRAMMAR
|
|
61
|
+
# is well formed -- terminal regexes compile, no terminal is zero-width, every
|
|
62
|
+
# %ignore name is defined -- which lark already established when it built this
|
|
63
|
+
# parser and which cannot change afterwards. It only ever raises or stays
|
|
64
|
+
# silent; it never influences which tokens come out.
|
|
65
|
+
_ERROR_LEXERS: "weakref.WeakKeyDictionary" = weakref.WeakKeyDictionary()
|
|
66
|
+
|
|
67
|
+
# Cached in place of a lexer for a parser this module cannot serve, so the
|
|
68
|
+
# decision is made once rather than re-attempted on every failed formula.
|
|
69
|
+
_UNAVAILABLE = object()
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _error_lexer(parser: Lark):
|
|
73
|
+
"""The cached, validation-free lexer for `parser`, or None if lark's
|
|
74
|
+
internals are not shaped the way this module expects.
|
|
75
|
+
|
|
76
|
+
BUILDING the lexer is allowed to fail -- lark could rename its classes, or
|
|
77
|
+
reject a grammar that only its own validation would have diagnosed -- and
|
|
78
|
+
falling back to ``parser.lex`` covers that. USING it is not: see
|
|
79
|
+
``lex_for_message``.
|
|
80
|
+
|
|
81
|
+
The cache READ and the cache WRITE are guarded separately from the
|
|
82
|
+
construction, and separately from each other. Not fussiness -- each guard
|
|
83
|
+
answers a different failure, and merging them breaks one of the two things
|
|
84
|
+
this function has to do at once:
|
|
85
|
+
|
|
86
|
+
* ``_ERROR_LEXERS`` is a WeakKeyDictionary, so it needs the parser to be
|
|
87
|
+
hashable and weak-referenceable. A lark that grew ``__slots__`` without
|
|
88
|
+
``__weakref__``, or an ``__eq__`` without ``__hash__``, makes the lookup
|
|
89
|
+
itself raise TypeError -- and unguarded, that TypeError comes out of
|
|
90
|
+
every ``NamingError`` in place of the message, turning a syntax error
|
|
91
|
+
into a crash.
|
|
92
|
+
* One try around the whole body would swallow the construction failure
|
|
93
|
+
before the ``_UNAVAILABLE`` write, so the failed construction would be
|
|
94
|
+
retried on every single failed formula instead of once. That is not
|
|
95
|
+
hypothetical: it is what the first attempt at this guard did, caught by
|
|
96
|
+
``test_a_grammar_lark_cannot_serve_is_only_attempted_once``.
|
|
97
|
+
|
|
98
|
+
Nothing in here lexes, so swallowing is safe throughout: the worst outcome
|
|
99
|
+
is the slow route, which is what the code did before this module existed.
|
|
100
|
+
"""
|
|
101
|
+
try:
|
|
102
|
+
cached = _ERROR_LEXERS.get(parser)
|
|
103
|
+
except Exception:
|
|
104
|
+
# The cache itself cannot hold this parser. Nothing to remember, so
|
|
105
|
+
# every failed parse takes the slow route -- but it gets its message.
|
|
106
|
+
return None
|
|
107
|
+
if cached is not None:
|
|
108
|
+
return None if cached is _UNAVAILABLE else cached
|
|
109
|
+
|
|
110
|
+
lexer = None
|
|
111
|
+
try:
|
|
112
|
+
if (_BasicLexer is not None and _LexerThread is not None
|
|
113
|
+
and hasattr(_LexerThread, "from_text")
|
|
114
|
+
# postlex: Lark.lex feeds the token stream through it. The MSFL
|
|
115
|
+
# grammars declare none, and reproducing that stage here would
|
|
116
|
+
# be guesswork, so such a parser goes the long way round.
|
|
117
|
+
and parser.options.postlex is None):
|
|
118
|
+
conf = copy(parser.lexer_conf)
|
|
119
|
+
conf.skip_validation = True
|
|
120
|
+
candidate = _BasicLexer(conf)
|
|
121
|
+
# Force the scanner NOW rather than on first use. It compiles the
|
|
122
|
+
# terminal regexes, which is the one step validation would have
|
|
123
|
+
# diagnosed with a readable LexError; doing it inside this try
|
|
124
|
+
# means a grammar lark can only lex the slow way falls back here,
|
|
125
|
+
# where falling back is free, instead of raising re.error out of
|
|
126
|
+
# an error message. It also moves the compile off the first
|
|
127
|
+
# failure's critical path.
|
|
128
|
+
candidate.scanner
|
|
129
|
+
lexer = candidate
|
|
130
|
+
except Exception:
|
|
131
|
+
lexer = None
|
|
132
|
+
|
|
133
|
+
try:
|
|
134
|
+
_ERROR_LEXERS[parser] = lexer if lexer is not None else _UNAVAILABLE
|
|
135
|
+
except Exception:
|
|
136
|
+
pass
|
|
137
|
+
return lexer
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def lex_for_message(parser: Lark, text: str) -> list:
|
|
141
|
+
"""Tokenize `text` the way the grammar does, for use in an error message.
|
|
142
|
+
|
|
143
|
+
Equivalent to ``list(parser.lex(text))`` but without lark's per-call
|
|
144
|
+
terminal-collision check -- see the comment above for why that check is
|
|
145
|
+
ruinous on a runtime-generated Unicode grammar.
|
|
146
|
+
|
|
147
|
+
Note what is NOT wrapped in a fallback: once a lexer exists, lexing runs
|
|
148
|
+
outside any try. Catching here would be actively harmful, because the
|
|
149
|
+
commonest failure is ``UnexpectedCharacters`` from `text` itself -- and
|
|
150
|
+
"catch it, then retry through ``parser.lex``" would reinstate the whole
|
|
151
|
+
cost this function exists to avoid, for exactly the malformed inputs that
|
|
152
|
+
make it matter, while ending in the same exception. ``parser.lex`` did not
|
|
153
|
+
swallow that exception either, so letting it through also keeps the
|
|
154
|
+
behaviour callers already had.
|
|
155
|
+
"""
|
|
156
|
+
lexer = _error_lexer(parser)
|
|
157
|
+
if lexer is None:
|
|
158
|
+
return list(parser.lex(text))
|
|
159
|
+
return list(_LexerThread.from_text(lexer, text).lex(None))
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
_SYMBOL_NAMES = {
|
|
163
|
+
"↔": "biconditional",
|
|
164
|
+
"→": "implication",
|
|
165
|
+
"∧": "conjunction",
|
|
166
|
+
"∨": "disjunction",
|
|
167
|
+
"⊕": "exclusive or",
|
|
168
|
+
"⊗": "strong conjunction",
|
|
169
|
+
"¬": "negation",
|
|
170
|
+
"∀": "universal quantifier",
|
|
171
|
+
"∃": "existential quantifier",
|
|
172
|
+
"≤": "less-than-or-equal",
|
|
173
|
+
"≥": "greater-than-or-equal",
|
|
174
|
+
"≠": "not-equal",
|
|
175
|
+
"=": "equality",
|
|
176
|
+
"<": "less-than",
|
|
177
|
+
">": "greater-than",
|
|
178
|
+
"+": "plus",
|
|
179
|
+
"-": "minus",
|
|
180
|
+
"*": "times",
|
|
181
|
+
"/": "division",
|
|
182
|
+
"(": "opening parenthesis",
|
|
183
|
+
")": "closing parenthesis",
|
|
184
|
+
"[": "opening bracket",
|
|
185
|
+
"]": "closing bracket",
|
|
186
|
+
",": "comma",
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
_NAMED_TOKENS = {
|
|
190
|
+
"PREDICATE": "predicate",
|
|
191
|
+
"NAME": "name/constant",
|
|
192
|
+
"VARIABLE": "variable",
|
|
193
|
+
"CONSTANT": "constant",
|
|
194
|
+
"QUOTED_NAME": "quoted constant",
|
|
195
|
+
"NUMBER": "number",
|
|
196
|
+
"SORT": "sort annotation",
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
_MIXING_INFO = {
|
|
200
|
+
"fol": (
|
|
201
|
+
{"∧", "∨", "⊕"},
|
|
202
|
+
"Cannot mix conjunction (∧), disjunction (∨), and exclusive or (⊕) without parentheses",
|
|
203
|
+
),
|
|
204
|
+
"msfol": (
|
|
205
|
+
{"∧", "∨"},
|
|
206
|
+
"Cannot mix conjunction (∧) and disjunction (∨) without parentheses",
|
|
207
|
+
),
|
|
208
|
+
"msfl": (
|
|
209
|
+
{"∧", "∨", "⊗", "⊕"},
|
|
210
|
+
"Cannot mix weak conjunction (∧), weak disjunction (∨), "
|
|
211
|
+
"strong conjunction (⊗), and strong disjunction (⊕) without parentheses",
|
|
212
|
+
),
|
|
213
|
+
"fl": (
|
|
214
|
+
{"∧", "∨", "⊗", "⊕"},
|
|
215
|
+
"Cannot mix weak conjunction (∧), weak disjunction (∨), "
|
|
216
|
+
"strong conjunction (⊗), and strong disjunction (⊕) without parentheses",
|
|
217
|
+
),
|
|
218
|
+
# Modal and second-order modes use the classical connectives at the same-level
|
|
219
|
+
# group (the modal/temporal/second-order operators bind tighter, like ¬).
|
|
220
|
+
"modal": (
|
|
221
|
+
{"∧", "∨", "⊕"},
|
|
222
|
+
"Cannot mix conjunction (∧), disjunction (∨), and exclusive or (⊕) without parentheses",
|
|
223
|
+
),
|
|
224
|
+
"so": (
|
|
225
|
+
{"∧", "∨", "⊕"},
|
|
226
|
+
"Cannot mix conjunction (∧), disjunction (∨), and exclusive or (⊕) without parentheses",
|
|
227
|
+
),
|
|
228
|
+
"dependence": (
|
|
229
|
+
{"∧", "∨"},
|
|
230
|
+
"Cannot mix conjunction (∧) and splitting disjunction (∨) without parentheses",
|
|
231
|
+
),
|
|
232
|
+
"linear": (
|
|
233
|
+
{"⊗", "&", "⊕"},
|
|
234
|
+
"Cannot mix tensor (⊗), with (&), and plus (⊕) without parentheses",
|
|
235
|
+
),
|
|
236
|
+
"lambek": (
|
|
237
|
+
{"•"},
|
|
238
|
+
"Parenthesise nested products (•) explicitly",
|
|
239
|
+
),
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def _build_pattern_index(parser: Lark) -> dict:
|
|
244
|
+
"""Map terminal name -> raw pattern string for every terminal in the grammar."""
|
|
245
|
+
index = {}
|
|
246
|
+
for term in parser.terminals:
|
|
247
|
+
pattern = term.pattern
|
|
248
|
+
index[term.name] = pattern.value if hasattr(pattern, "value") else str(pattern)
|
|
249
|
+
return index
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def _display_name(token_type: str, patterns: dict) -> str:
|
|
253
|
+
"""Resolve a terminal name to a human-readable label.
|
|
254
|
+
|
|
255
|
+
Named tokens use a fixed label; symbol tokens are resolved through their
|
|
256
|
+
pattern; anything unknown falls back to the raw terminal name.
|
|
257
|
+
"""
|
|
258
|
+
if token_type in _NAMED_TOKENS:
|
|
259
|
+
return _NAMED_TOKENS[token_type]
|
|
260
|
+
pattern = patterns.get(token_type)
|
|
261
|
+
if pattern in _SYMBOL_NAMES:
|
|
262
|
+
return _SYMBOL_NAMES[pattern]
|
|
263
|
+
return token_type
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _is_structural(token_type: str) -> bool:
|
|
267
|
+
"""True if the token is a symbol/operator rather than a name-like token."""
|
|
268
|
+
return token_type not in _NAMED_TOKENS
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def _format_expected(expected, patterns: dict) -> str:
|
|
272
|
+
"""Render a set of expected terminal names as a sorted, deduplicated label list."""
|
|
273
|
+
labels = []
|
|
274
|
+
seen = set()
|
|
275
|
+
for token_type in expected or ():
|
|
276
|
+
label = _display_name(token_type, patterns)
|
|
277
|
+
if label not in seen:
|
|
278
|
+
seen.add(label)
|
|
279
|
+
labels.append(label)
|
|
280
|
+
return ", ".join(sorted(labels)) if labels else "a valid token"
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
class NamingError(UnexpectedCharacters):
|
|
284
|
+
"""Human-readable wrapper around a lexer-level UnexpectedCharacters failure.
|
|
285
|
+
|
|
286
|
+
Token names are resolved through the grammar's terminal patterns rather
|
|
287
|
+
than hard-coded anonymous-token numbers, so messages stay correct when the
|
|
288
|
+
grammar changes.
|
|
289
|
+
"""
|
|
290
|
+
|
|
291
|
+
def __init__(self, parser: Lark, original_exception: UnexpectedCharacters, formula: str, mode: str = "fol"):
|
|
292
|
+
self._patterns = _build_pattern_index(parser)
|
|
293
|
+
self._mode = mode
|
|
294
|
+
|
|
295
|
+
pos = original_exception.pos_in_stream
|
|
296
|
+
prefix = formula[:pos] if pos is not None and pos >= 0 else formula
|
|
297
|
+
tokens = lex_for_message(parser, prefix)
|
|
298
|
+
last_token = tokens[-1] if tokens else None
|
|
299
|
+
|
|
300
|
+
if last_token is None:
|
|
301
|
+
message = (
|
|
302
|
+
f"SYNTAX_ERROR: Unexpected character "
|
|
303
|
+
f"'{original_exception.char}' at position {original_exception.column}"
|
|
304
|
+
)
|
|
305
|
+
else:
|
|
306
|
+
message = self._build_message(last_token, original_exception)
|
|
307
|
+
|
|
308
|
+
self.__dict__.update(original_exception.__dict__)
|
|
309
|
+
self.args = (message,)
|
|
310
|
+
|
|
311
|
+
def _build_message(self, last_token, exc: UnexpectedCharacters) -> str:
|
|
312
|
+
"""Compose the final error message for a lexer-level failure."""
|
|
313
|
+
display = _display_name(last_token.type, self._patterns)
|
|
314
|
+
mixing_symbols, mixing_hint = _MIXING_INFO.get(self._mode, _MIXING_INFO["fol"])
|
|
315
|
+
mixes = exc.char in mixing_symbols
|
|
316
|
+
|
|
317
|
+
# A same-level connective is never a NAMING failure, even when the token
|
|
318
|
+
# in front of it is name-like: in 'A ∧ B ∨ C' the predicate 'B' is
|
|
319
|
+
# perfectly well formed and it is the MIX that the grammar refuses.
|
|
320
|
+
# Reporting it as "Invalid predicate 'B' … Expected pattern: [A-Z]…"
|
|
321
|
+
# sends the reader (or a generating model) off renaming a good name
|
|
322
|
+
# instead of adding the brackets, so the mixing case takes the
|
|
323
|
+
# structural wording — which carries the hint — in both branches.
|
|
324
|
+
if _is_structural(last_token.type) or mixes:
|
|
325
|
+
message = (
|
|
326
|
+
f"SYNTAX_ERROR: Unexpected character '{exc.char}' at position "
|
|
327
|
+
f"{exc.column} after {display} '{last_token.value}'"
|
|
328
|
+
)
|
|
329
|
+
if mixes:
|
|
330
|
+
message += f". Hint: {mixing_hint}"
|
|
331
|
+
return message
|
|
332
|
+
|
|
333
|
+
if last_token.type == "QUOTED_NAME":
|
|
334
|
+
after_quoted = self._after_quoted_constant(last_token, exc)
|
|
335
|
+
if after_quoted is not None:
|
|
336
|
+
return after_quoted
|
|
337
|
+
|
|
338
|
+
message = (
|
|
339
|
+
f"SYNTAX_ERROR: Invalid {display} '{last_token.value}' - "
|
|
340
|
+
f"unexpected character '{exc.char}' at position {exc.column}"
|
|
341
|
+
)
|
|
342
|
+
# PREDICATE/NAME/CONSTANT/VARIABLE/SORT are generated at runtime from
|
|
343
|
+
# the interpreter's Unicode tables (see fol/_identifiers.py) — their
|
|
344
|
+
# raw pattern text is several kilobytes of \uXXXX-\uYYYY ranges, which
|
|
345
|
+
# is not a message a human (or a model reading the error to retry)
|
|
346
|
+
# can act on, so those five show a short hand-written description
|
|
347
|
+
# instead of the pattern this module resolved from the grammar. So does
|
|
348
|
+
# QUOTED_NAME: its pattern is short, but a character class of escapes
|
|
349
|
+
# does not say "a quote is written \' inside".
|
|
350
|
+
human = HUMAN_READABLE_PATTERNS.get(last_token.type)
|
|
351
|
+
if human:
|
|
352
|
+
message += f". Expected pattern: {human}"
|
|
353
|
+
else:
|
|
354
|
+
pattern = self._patterns.get(last_token.type)
|
|
355
|
+
if pattern:
|
|
356
|
+
message += f". Expected pattern: {pattern}"
|
|
357
|
+
return message
|
|
358
|
+
|
|
359
|
+
def _after_quoted_constant(self, last_token, exc: UnexpectedCharacters) -> Optional[str]:
|
|
360
|
+
"""The message for the two things that go wrong AFTER a quoted constant, else ``None``.
|
|
361
|
+
|
|
362
|
+
A quoted constant the lexer matched is complete, so "Invalid quoted constant"
|
|
363
|
+
would blame a good name where the mistake is what follows it:
|
|
364
|
+
|
|
365
|
+
* an opening parenthesis: the writer meant a predicate or a function with a
|
|
366
|
+
quoted name (TPTP has one, ``'foo bar'(a)``), and this grammar quotes
|
|
367
|
+
constants only;
|
|
368
|
+
* in a many-sorted grammar, anything but the sort: a constant carries its sort
|
|
369
|
+
there (``'k2':Mountain``), and the bare ``'k2'`` is what is missing one.
|
|
370
|
+
|
|
371
|
+
Any other character keeps the general wording, which describes the shape of
|
|
372
|
+
the token: that is the case of a stray apostrophe (``P('a'b)``).
|
|
373
|
+
"""
|
|
374
|
+
where = (f"SYNTAX_ERROR: Unexpected character '{exc.char}' at position {exc.column} "
|
|
375
|
+
f"after the quoted constant {last_token.value}")
|
|
376
|
+
if exc.char == "(":
|
|
377
|
+
return (f"{where}. A name in quotes is a constant and takes no arguments; a "
|
|
378
|
+
f"predicate or function name has no quoted form")
|
|
379
|
+
if "SORT" in self._patterns and exc.char != ":":
|
|
380
|
+
return (f"{where}. In a many-sorted formula a constant carries its sort: write "
|
|
381
|
+
f"{last_token.value}:Sort")
|
|
382
|
+
return None
|
|
383
|
+
|
|
384
|
+
def __str__(self):
|
|
385
|
+
return self.args[0]
|
|
386
|
+
|
|
387
|
+
|
|
388
|
+
class ParsingError(ParseError):
|
|
389
|
+
"""Human-readable wrapper around parser-level failures.
|
|
390
|
+
|
|
391
|
+
Handles both UnexpectedToken (a valid token in an invalid position) and
|
|
392
|
+
UnexpectedEOF (the formula ended before it was complete). The expected-token
|
|
393
|
+
set is rendered through the same pattern-based resolution as NamingError.
|
|
394
|
+
"""
|
|
395
|
+
|
|
396
|
+
def __init__(self, parser: Lark, original_exception, formula: str, mode: str = "fol"):
|
|
397
|
+
self._patterns = _build_pattern_index(parser)
|
|
398
|
+
expected_str = _format_expected(getattr(original_exception, "expected", None), self._patterns)
|
|
399
|
+
|
|
400
|
+
if isinstance(original_exception, UnexpectedEOF):
|
|
401
|
+
message = (
|
|
402
|
+
f"SYNTAX_ERROR: Incomplete formula - the input ended unexpectedly. "
|
|
403
|
+
f"Expected: {expected_str}"
|
|
404
|
+
)
|
|
405
|
+
else:
|
|
406
|
+
token = original_exception.token
|
|
407
|
+
display = _display_name(token.type, self._patterns)
|
|
408
|
+
column = getattr(original_exception, "column", None)
|
|
409
|
+
where = f" at position {column}" if column not in (None, -1) else ""
|
|
410
|
+
message = (
|
|
411
|
+
f"SYNTAX_ERROR: Unexpected {display} '{token.value}'{where}. "
|
|
412
|
+
f"Expected: {expected_str}"
|
|
413
|
+
)
|
|
414
|
+
mixing_symbols, mixing_hint = _MIXING_INFO.get(mode, _MIXING_INFO["fol"])
|
|
415
|
+
if str(token.value) in mixing_symbols:
|
|
416
|
+
message += f". Hint: {mixing_hint}"
|
|
417
|
+
|
|
418
|
+
self.__dict__.update(original_exception.__dict__)
|
|
419
|
+
self.args = (message,)
|
|
420
|
+
|
|
421
|
+
def __str__(self):
|
|
422
|
+
return self.args[0]
|
|
@@ -0,0 +1,241 @@
|
|
|
1
|
+
"""Public re-export hub for all AST node classes and utilities.
|
|
2
|
+
|
|
3
|
+
Classical FOL definitions live in _fol_nodes.py.
|
|
4
|
+
MSFL extension (sorted quantifiers/constants, Łukasiewicz operators, to_fol) lives in _msfl_nodes.py.
|
|
5
|
+
|
|
6
|
+
STABLE PUBLIC API
|
|
7
|
+
------------------
|
|
8
|
+
Two things exported from this module are STABLE PUBLIC API, in the sense
|
|
9
|
+
that a downstream consumer is meant to build directly on them rather than
|
|
10
|
+
reaching past them into the modules underneath (``_fol_nodes.py``,
|
|
11
|
+
``_msfl_nodes.py``, …, all underscore-prefixed precisely because THEY are
|
|
12
|
+
not the contract):
|
|
13
|
+
|
|
14
|
+
* every node class's CONSTRUCTOR — ``Variable``, ``Constant``, ``Number``,
|
|
15
|
+
``Function``, ``Atom``, ``Not``, ``And``, ``Or``, ``Xor``, ``Implies``,
|
|
16
|
+
``Iff``, ``Quantifier``, ``Count``, ``Measure``, ``Cardinality``,
|
|
17
|
+
``Contrast``, and every MSFL/modal/hybrid/team/linear/lambek/
|
|
18
|
+
second-order class re-exported below — meaning field names, field
|
|
19
|
+
order, and what a positional or keyword argument means;
|
|
20
|
+
* ``Node.to_unicode_str()``, meaning both that it renders a parseable
|
|
21
|
+
Unicode formula string, and that the string it renders parses back
|
|
22
|
+
(via the matching ``MSFLParser`` mode) to a structurally equal node —
|
|
23
|
+
the roundtrip guarantee documented on ``to_unicode_str`` itself and
|
|
24
|
+
exercised by the FOL-fragment roundtrip test suite. A constant whose name
|
|
25
|
+
does not read back as that constant when written bare (``k2`` is a
|
|
26
|
+
variable, ``Alice`` a predicate, ``G-910`` no term) is written in single
|
|
27
|
+
quotes, ``'k2'``, so that the text reads back for every constant that has
|
|
28
|
+
a text; a constant that reads back bare (``socrates``) is written as it
|
|
29
|
+
always was.
|
|
30
|
+
|
|
31
|
+
A consumer that builds nodes by calling these constructors directly and
|
|
32
|
+
serialises them back to text via ``to_unicode_str`` — the way a mutation-
|
|
33
|
+
search loop assembling candidate formulas and shipping them across a
|
|
34
|
+
process boundary would — is standing on committed API, not on an
|
|
35
|
+
implementation detail that happens to work today.
|
|
36
|
+
|
|
37
|
+
What "stable" commits this kit to: a node class is not renamed, its
|
|
38
|
+
dataclass fields are not reordered or reinterpreted, and ``to_unicode_str``'s
|
|
39
|
+
grammar does not change in a way that breaks the roundtrip for an existing
|
|
40
|
+
node shape, without that change being called out explicitly in
|
|
41
|
+
``CHANGELOG.md`` — never folded silently into an unrelated change. This
|
|
42
|
+
project is pre-1.0 (see ``CHANGELOG.md``'s header: "a minor release MAY
|
|
43
|
+
contain breaking changes"), so a break is not impossible — but for these
|
|
44
|
+
names it is never an accident. ``Node.__eq__`` / ``__hash__`` / ``repr`` /
|
|
45
|
+
``to_dict`` / ``from_dict`` are part of the same commitment: unchanged for
|
|
46
|
+
every node already registered in ``NODE_CLASSES``.
|
|
47
|
+
|
|
48
|
+
What it does NOT commit to: anything not re-exported here or in
|
|
49
|
+
``fol/__init__.py``'s ``__all__`` — internal helpers, the exact
|
|
50
|
+
``_child_nodes()``/``map_children`` traversal order, or any
|
|
51
|
+
underscore-prefixed name in any ``_*_nodes.py`` module. The separate, PATH
|
|
52
|
+
convention ``replace_at``/``node_at``/``fol.spans.traverse`` agree on
|
|
53
|
+
(``_child_nodes()`` order, except a ``Quantifier``'s bound variable is
|
|
54
|
+
excluded — see ``_fol_nodes.py``'s "Public tree editing" section and
|
|
55
|
+
``fol/spans.py``'s module docstring) is its own narrower commitment, scoped
|
|
56
|
+
to those three functions.
|
|
57
|
+
"""
|
|
58
|
+
|
|
59
|
+
from ._fol_nodes import (
|
|
60
|
+
Z3Env,
|
|
61
|
+
Node,
|
|
62
|
+
Variable, Constant, Number, Function,
|
|
63
|
+
Atom, Not, And, Or, Xor, Implies, Iff, Quantifier,
|
|
64
|
+
Count, Measure, Cardinality, Contrast,
|
|
65
|
+
NODE_CLASSES,
|
|
66
|
+
FOLTransformer,
|
|
67
|
+
node_at, replace_at,
|
|
68
|
+
)
|
|
69
|
+
from ._msfl_nodes import (
|
|
70
|
+
SortedQuantifier, SortedConstant,
|
|
71
|
+
SortedCount, SortedCardinality,
|
|
72
|
+
WeakConjunction, WeakDisjunction,
|
|
73
|
+
StrongConjunction, StrongDisjunction,
|
|
74
|
+
LukNegation, LukImplication, LukEquivalence,
|
|
75
|
+
LambdaVar, Lambda, Application,
|
|
76
|
+
free_variables,
|
|
77
|
+
substitute, beta_reduce, ReductionLimitError,
|
|
78
|
+
eta_reduce, beta_eta_normalize,
|
|
79
|
+
resolve_lambda_scope,
|
|
80
|
+
to_fol,
|
|
81
|
+
nonempty_sort_axioms,
|
|
82
|
+
sort_membership_axioms,
|
|
83
|
+
sort_axioms,
|
|
84
|
+
subsort_axioms,
|
|
85
|
+
signature_axioms,
|
|
86
|
+
)
|
|
87
|
+
from ._modal_nodes import (
|
|
88
|
+
Box, Diamond, Knows, Believes, Says, Wants,
|
|
89
|
+
EverybodyKnows, DistributedKnowledge, CommonKnowledge,
|
|
90
|
+
Always, Eventually, Next, Until,
|
|
91
|
+
Historically, Once, Previous, Since,
|
|
92
|
+
Obligatory, Permitted,
|
|
93
|
+
Would, Might,
|
|
94
|
+
Announce, AnnounceDiamond,
|
|
95
|
+
)
|
|
96
|
+
from ._so_nodes import SecondOrderQuantifier
|
|
97
|
+
from ._hybrid_nodes import Nominal, At, Down
|
|
98
|
+
from ._team_nodes import Dependence, SlashedExists
|
|
99
|
+
from ._linear_nodes import (
|
|
100
|
+
Tensor, With, OPlus, LinearImplies, OfCourse, One, Top, Zero,
|
|
101
|
+
)
|
|
102
|
+
from ._lambek_nodes import Product, Under, Over
|
|
103
|
+
# Imported LAST: _ho_nodes clones the modal and second-order parser
|
|
104
|
+
# registrations into the two third-order grammar modes, so every module
|
|
105
|
+
# that registers an operator for those modes -- _hybrid_nodes included --
|
|
106
|
+
# has to have run first.
|
|
107
|
+
from ._ho_nodes import (
|
|
108
|
+
PredicateTerm, Signatures, analyse_signatures, MixedSlotError, NestedPropertySlotError,
|
|
109
|
+
_clone_parser_ops,
|
|
110
|
+
)
|
|
111
|
+
# PARSER_OPS/ParserOp/parser_ops_for_mode: needed below by
|
|
112
|
+
# _clone_parser_ops_sorted, the exclusion-aware sibling of _ho_nodes'
|
|
113
|
+
# _clone_parser_ops that assembles the two new SORTED+modal/second-order
|
|
114
|
+
# grammar modes (see that function's docstring for why plain _clone_parser_ops
|
|
115
|
+
# is not enough here).
|
|
116
|
+
from ._fol_nodes import PARSER_OPS, ParserOp, parser_ops_for_mode
|
|
117
|
+
|
|
118
|
+
# The two third-order grammar modes are their base modes' operator sets over a
|
|
119
|
+
# widened argument layer, so they are assembled by CLONING rather than by
|
|
120
|
+
# re-registering ~40 operators that would then drift. It happens here, and
|
|
121
|
+
# happens last, because it can only be correct once every module that registers
|
|
122
|
+
# an operator for "modal" or "second_order" has been imported -- which, at this
|
|
123
|
+
# point in this file, they all have.
|
|
124
|
+
_clone_parser_ops("third_order", ["second_order"])
|
|
125
|
+
_clone_parser_ops("third_order_modal", ["modal", "second_order"])
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
# =========================
|
|
129
|
+
# many_sorted + modal / second_order grammar modes (C3)
|
|
130
|
+
# =========================
|
|
131
|
+
#
|
|
132
|
+
# "modal_sorted" (MSFLParser(modal=True, many_sorted=True)) and "so_sorted"
|
|
133
|
+
# (MSFLParser(second_order=True, many_sorted=True)) are assembled the same
|
|
134
|
+
# CLONING way the third-order modes above are -- but plain _clone_parser_ops
|
|
135
|
+
# is not quite enough here, for a reason the third-order clones never hit:
|
|
136
|
+
# "modal" and "second_order" each register the UNSORTED individual quantifier
|
|
137
|
+
# (Quantifier, rule alias "quantifier_") and the unsorted counting quantifier
|
|
138
|
+
# (Count, "count_"), while "msfol" registers the SORTED equivalents
|
|
139
|
+
# (SortedQuantifier "sorted_quantifier_", SortedCount "sorted_count_") under
|
|
140
|
+
# DIFFERENT rule aliases / grammar fragments. _clone_parser_ops's dedup key is
|
|
141
|
+
# (level, rule_alias, grammar, only_name) -- since the sorted and unsorted
|
|
142
|
+
# forms differ on every one of those, a bare clone of ["modal", "msfol"]
|
|
143
|
+
# would keep BOTH, so ``MSFLParser(modal=True, many_sorted=True)`` would
|
|
144
|
+
# accept an UNSORTED ``∀x P(x)`` right alongside ``∀x:S P(x)`` -- silently
|
|
145
|
+
# reintroducing the unsorted quantifier many_sorted is supposed to forbid
|
|
146
|
+
# (exactly like plain "msfol" forbids it today). _clone_parser_ops_sorted
|
|
147
|
+
# below is _ho_nodes._clone_parser_ops with one addition: ops whose rule_alias
|
|
148
|
+
# names an unsorted binder are skipped, so the sorted mode ends up with
|
|
149
|
+
# EXACTLY "msfol"'s quantifier/count/constant/cardinality forms plus every
|
|
150
|
+
# modal / second-order operator, and nothing double-registered (the classical
|
|
151
|
+
# connectives and Contrast register identically -- same level/rule_alias/
|
|
152
|
+
# grammar/only_name -- for "modal"/"second_order" and "msfol", so the
|
|
153
|
+
# ordinary dedup already collapses those to one copy each).
|
|
154
|
+
_UNSORTED_BINDER_ALIASES = frozenset({"quantifier_", "count_"})
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _clone_parser_ops_sorted(target: str, sources) -> None:
|
|
158
|
+
"""Like ``_ho_nodes._clone_parser_ops(target, sources)``, but never clones
|
|
159
|
+
an unsorted quantifier/counting-quantifier binding (see the module comment
|
|
160
|
+
above). ``sources`` should list the SORTED source mode ("msfol") before
|
|
161
|
+
the modal/second-order one, so a genuine grammar conflict — should one
|
|
162
|
+
ever appear — is reported against the sorted form's own shape first;
|
|
163
|
+
today no such conflict exists, since every non-binder op the two source
|
|
164
|
+
modes share registers identically and is deduped as usual.
|
|
165
|
+
"""
|
|
166
|
+
seen = set()
|
|
167
|
+
for source in sources:
|
|
168
|
+
for op in parser_ops_for_mode(source):
|
|
169
|
+
if op.rule_alias in _UNSORTED_BINDER_ALIASES:
|
|
170
|
+
continue
|
|
171
|
+
key = (op.level, op.rule_alias, op.grammar, op.only_name)
|
|
172
|
+
if key in seen:
|
|
173
|
+
continue
|
|
174
|
+
seen.add(key)
|
|
175
|
+
PARSER_OPS.append(ParserOp(
|
|
176
|
+
target, op.level, op.terminal_name, op.terminal_def,
|
|
177
|
+
op.grammar, op.rule_alias, op.transform, op.node_class,
|
|
178
|
+
op.only_name))
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
_clone_parser_ops_sorted("modal_sorted", ["msfol", "modal"])
|
|
182
|
+
_clone_parser_ops_sorted("so_sorted", ["msfol", "second_order"])
|
|
183
|
+
|
|
184
|
+
# build_grammar (fol/_fol_nodes.py) also needs two pieces of per-mode
|
|
185
|
+
# configuration that are NOT operator-specific and therefore not covered by
|
|
186
|
+
# PARSER_OPS cloning above: whether SORT is a recognised terminal / bare
|
|
187
|
+
# constants must carry a sort annotation (_SORTED_MODES), and the terminal
|
|
188
|
+
# import list (_MODE_TERMINAL_IMPORTS, indexed with `[mode]` -- a missing key
|
|
189
|
+
# is a hard KeyError). Both live in fol/_fol_nodes.py, which this change does
|
|
190
|
+
# not own/edit; they are extended here, at runtime, the same way
|
|
191
|
+
# _clone_parser_ops_sorted above extends PARSER_OPS (another registry that
|
|
192
|
+
# also lives in _fol_nodes.py) -- module-level registries, not the module's
|
|
193
|
+
# own source, so this is additive rather than a hack around ownership. Every
|
|
194
|
+
# other mode-keyed dict build_grammar reads (_MODE_TERM_EXTRA,
|
|
195
|
+
# _MODE_ATOM_ARGS, _MODE_ATOM_EXTRA) is read with ``.get(mode, default)``, so
|
|
196
|
+
# the two sorted modes correctly fall back to "no extra term form" / "plain
|
|
197
|
+
# termlist" / "no extra atom form" without needing an entry.
|
|
198
|
+
from . import _fol_nodes as _fn # noqa: E402 (after the registrations above)
|
|
199
|
+
|
|
200
|
+
_fn._SORTED_MODES = _fn._SORTED_MODES | {"modal_sorted", "so_sorted"}
|
|
201
|
+
_fn._MODE_TERMINAL_IMPORTS.setdefault("modal_sorted", _fn._MODE_TERMINAL_IMPORTS["modal"])
|
|
202
|
+
_fn._MODE_TERMINAL_IMPORTS.setdefault("so_sorted", _fn._MODE_TERMINAL_IMPORTS["second_order"])
|
|
203
|
+
|
|
204
|
+
__all__ = [
|
|
205
|
+
"Z3Env",
|
|
206
|
+
"Node",
|
|
207
|
+
"Variable", "Constant", "Number", "Function",
|
|
208
|
+
"Atom", "Not", "And", "Or", "Xor", "Implies", "Iff", "Quantifier",
|
|
209
|
+
"Count", "Measure", "Cardinality", "Contrast",
|
|
210
|
+
"NODE_CLASSES",
|
|
211
|
+
"FOLTransformer",
|
|
212
|
+
"node_at", "replace_at",
|
|
213
|
+
"SortedQuantifier", "SortedConstant",
|
|
214
|
+
"SortedCount", "SortedCardinality",
|
|
215
|
+
"Nominal", "At", "Down",
|
|
216
|
+
"Dependence", "SlashedExists",
|
|
217
|
+
"Tensor", "With", "OPlus", "LinearImplies", "OfCourse", "One", "Top", "Zero",
|
|
218
|
+
"Product", "Under", "Over",
|
|
219
|
+
"WeakConjunction", "WeakDisjunction",
|
|
220
|
+
"StrongConjunction", "StrongDisjunction",
|
|
221
|
+
"LukNegation", "LukImplication", "LukEquivalence",
|
|
222
|
+
"LambdaVar", "Lambda", "Application",
|
|
223
|
+
"Box", "Diamond", "Knows", "Believes", "Says", "Wants",
|
|
224
|
+
"EverybodyKnows", "DistributedKnowledge", "CommonKnowledge",
|
|
225
|
+
"Always", "Eventually", "Next", "Until",
|
|
226
|
+
"Historically", "Once", "Previous", "Since",
|
|
227
|
+
"Obligatory", "Permitted",
|
|
228
|
+
"Would", "Might",
|
|
229
|
+
"SecondOrderQuantifier",
|
|
230
|
+
"PredicateTerm", "Signatures", "analyse_signatures", "MixedSlotError", "NestedPropertySlotError",
|
|
231
|
+
"free_variables",
|
|
232
|
+
"substitute", "beta_reduce", "ReductionLimitError",
|
|
233
|
+
"eta_reduce", "beta_eta_normalize",
|
|
234
|
+
"resolve_lambda_scope",
|
|
235
|
+
"to_fol",
|
|
236
|
+
"nonempty_sort_axioms",
|
|
237
|
+
"sort_membership_axioms",
|
|
238
|
+
"sort_axioms",
|
|
239
|
+
"subsort_axioms",
|
|
240
|
+
"signature_axioms",
|
|
241
|
+
]
|