unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,1091 @@
|
|
|
1
|
+
"""Runtime-generated identifier-terminal patterns for the FOL grammar.
|
|
2
|
+
|
|
3
|
+
WHY GENERATED, NOT FROZEN INTO A .lark FILE
|
|
4
|
+
---------------------------------------------
|
|
5
|
+
The predicate/name/constant/variable terminals used to be one hand-written
|
|
6
|
+
ASCII regex each (``PREDICATE: /[A-Z][a-zA-Z0-9]*/`` and siblings), which is
|
|
7
|
+
exactly why they rejected a FOLIO gold formula's ``LostTo(x, świątek)`` or
|
|
8
|
+
``Hosted(beijing, 2008SummerOlympics)`` outright: nothing outside `A-Za-z0-9`
|
|
9
|
+
was a letter or digit as far as the grammar was concerned. The fix is not a
|
|
10
|
+
bigger hand-written regex — pasting in, say, the Latin-1 Supplement and Latin
|
|
11
|
+
Extended-A block boundaries would still reject Cyrillic, CJK, Devanagari, and
|
|
12
|
+
everything else, and worse, would go stale the moment Unicode adds a new
|
|
13
|
+
script or a codepoint's General_Category changes, silently drifting from
|
|
14
|
+
whatever Python interpreter (3.10+) actually runs the kit. There is already a
|
|
15
|
+
correct, always-current answer to "is this codepoint a letter, and is it
|
|
16
|
+
uppercase" sitting in the interpreter: ``str.isalpha()`` / ``str.isupper()``.
|
|
17
|
+
So the character classes below are computed BY SCANNING those tables once,
|
|
18
|
+
in Python, at first use in a process (behind ``functools.lru_cache`` — the
|
|
19
|
+
scan costs well under a tenth of a second, which is a one-time-per-process
|
|
20
|
+
cost this module pays exactly once, not a per-``MSFLParser()`` or per-parse
|
|
21
|
+
cost), and handed to Lark as the ordinary compiled regex text it already
|
|
22
|
+
expects. A codepoint range is never typed into source; it is read off the
|
|
23
|
+
interpreter that will also be doing the matching.
|
|
24
|
+
|
|
25
|
+
WHY GREEK AND OHM ARE CARVED OUT OF EVERY LETTER CLASS
|
|
26
|
+
---------------------------------------------------------
|
|
27
|
+
``λ`` is the LAMBDA terminal, ``μ`` is the measure-term operator, and the
|
|
28
|
+
plain lowercase Greek letters (``αβγδεζηθικνξοπρστυφχψω``) are already the
|
|
29
|
+
CONSTANT terminal's second, non-``c_`` alternative — all three predate this
|
|
30
|
+
module and are matched as literal/fixed patterns elsewhere in the grammar.
|
|
31
|
+
A "just widen the letter classes to every Unicode letter" pass would swallow
|
|
32
|
+
all of that: ``λ`` and ``μ`` would stop being operators and start being
|
|
33
|
+
ordinary lowercase letters eligible to open a NAME or VARIABLE token, and the
|
|
34
|
+
Greek CONSTANT alternative would become redundant with (and shadowed by, or
|
|
35
|
+
racing against) a NAME/VARIABLE token now built from the exact same
|
|
36
|
+
characters. So Greek and Coptic (U+0370-U+03FF), Greek Extended
|
|
37
|
+
(U+1F00-U+1FFF, the block holding the accented/breathed forms), and U+2126
|
|
38
|
+
OHM SIGN (Unicode's own compatibility duplicate of Greek OMEGA, canonically
|
|
39
|
+
equivalent to U+03A9) are excluded from every letter this module recognises
|
|
40
|
+
— uppercase, lowercase, and the combining-mark continuation class alike.
|
|
41
|
+
Greek identifiers do not get a UPGRADE from this module; they keep exactly
|
|
42
|
+
the behaviour they had before it existed.
|
|
43
|
+
|
|
44
|
+
WHY THE FIRST CHARACTER DECIDES PREDICATE VS. TERM, AND WHAT THAT MEANS FOR
|
|
45
|
+
SCRIPTS WITHOUT CASE
|
|
46
|
+
-----------------------------------------------------------------------------
|
|
47
|
+
The kit's grammar has always used capitalisation to tell a predicate from a
|
|
48
|
+
term at the lexer level — no keyword, no sigil, just "the first letter is
|
|
49
|
+
uppercase" — so widening the alphabet has to widen that same signal rather
|
|
50
|
+
than replace it. Concretely: a token whose first character satisfies
|
|
51
|
+
``str.isupper()`` classifies as PREDICATE; every other first LETTER
|
|
52
|
+
classifies as term-valued (NAME, CONSTANT, or VARIABLE, depending on length
|
|
53
|
+
and the ``c_``/Greek forms). Most scripts have no case distinction at all
|
|
54
|
+
(CJK, Arabic, Hebrew, Devanagari, Thai, ...) — for every one of them,
|
|
55
|
+
``str.isupper()`` is simply always False, so an identifier in such a script
|
|
56
|
+
is always term-valued and can never head an atom by itself. That is not an
|
|
57
|
+
oversight this module works around; it is the literal reading of the
|
|
58
|
+
existing rule extended honestly to alphabets it was never tested against:
|
|
59
|
+
predicate-hood is signalled by a case distinction, and where a script draws
|
|
60
|
+
no such distinction, there is nothing to signal it with, so term position is
|
|
61
|
+
what a bare identifier in that script gets. (A caseless-script PREDICATE is
|
|
62
|
+
still reachable the way it always was: spell the atom with a Latin
|
|
63
|
+
uppercase-first name.)
|
|
64
|
+
|
|
65
|
+
WHY A DIGIT-LEADING IDENTIFIER IS ALWAYS TERM-VALUED, NEVER A PREDICATE
|
|
66
|
+
---------------------------------------------------------------------------
|
|
67
|
+
``2008SummerOlympics`` has to lex as SOMETHING for
|
|
68
|
+
``Hosted(beijing, 2008SummerOlympics)`` to parse, but it cannot become a
|
|
69
|
+
second numeric terminal (that would make ``NUMBER`` and the new form
|
|
70
|
+
ambiguous over every plain integer) and it cannot become eligible for
|
|
71
|
+
PREDICATE position (nothing in the classical FOL literature, and nothing
|
|
72
|
+
elsewhere in this grammar, lets an atom's head start with a digit — and
|
|
73
|
+
doing so would need its own case-style signal for predicate-hood, which a
|
|
74
|
+
digit doesn't carry). So a digit-leading identifier is folded into NAME:
|
|
75
|
+
one-or-more ASCII digits, then a letter (of either case-class), then the
|
|
76
|
+
ordinary NAME continuation. It reuses NAME's own transform (a bare NAME
|
|
77
|
+
token becomes a ``Constant``; ``NAME "(" ... ")"`` becomes a ``Function``
|
|
78
|
+
head), so ``2008SummerOlympics`` used bare is a ``Constant`` and
|
|
79
|
+
``2008SummerOlympics(x)`` would be a function call — never an atom. ASCII
|
|
80
|
+
digits only, deliberately: NUMBER is untouched by this whole module (``2008``
|
|
81
|
+
still lexes as NUMBER, ``2.5`` still lexes as NUMBER), so the digit run here
|
|
82
|
+
is exactly the digits NUMBER itself would have recognised — the difference
|
|
83
|
+
is solely the mandatory trailing letter that pulls the token out of NUMBER's
|
|
84
|
+
territory and into NAME's.
|
|
85
|
+
|
|
86
|
+
WHY CONSTANT KEEPS ITS PRIORITY, AND WHY THAT PRIORITY NOW MATTERS MORE
|
|
87
|
+
---------------------------------------------------------------------------
|
|
88
|
+
Before this module existed, CONSTANT (``c_...`` or a Greek run) and NAME
|
|
89
|
+
(``[a-z]...``) could never both match the same span: NAME never contained an
|
|
90
|
+
underscore, so ``c_alpha`` was CONSTANT-only ground. Widening NAME to accept
|
|
91
|
+
underscores as a continuation character (needed for ``dani_Shapiro`` and
|
|
92
|
+
``family_History``) removes that separation — ``c_alpha`` now also matches
|
|
93
|
+
NAME's alpha-leading form in full (``c`` is a lowercase letter, ``_alpha``
|
|
94
|
+
is legal NAME continuation with one more letter further on). Lark resolves
|
|
95
|
+
same-span multi-terminal ambiguity by priority, and CONSTANT was already
|
|
96
|
+
declared at priority 3 against NAME's 2 (``CONSTANT.3`` / ``NAME.2`` in the
|
|
97
|
+
grammar, both unchanged by this module), so ``c_alpha`` still lexes as
|
|
98
|
+
CONSTANT. The node is ``Constant("c_alpha")`` on either path: the ``const_``
|
|
99
|
+
transform keeps the mark as part of the name. No BARE text reads as a
|
|
100
|
+
``Constant`` whose name is a variable token (``k2`` is a variable); such a
|
|
101
|
+
constant is written in quotes, ``'k2'`` (see the QUOTED_NAME section below).
|
|
102
|
+
The two terminals now genuinely overlap where they never used to, so that
|
|
103
|
+
priority ordering has gone from "never exercised" to "load-bearing", and is
|
|
104
|
+
exercised by an explicit regression test (see
|
|
105
|
+
``tests/test_identifier_widening.py``) rather than left to be an accident of
|
|
106
|
+
how NAME happened to be spelled.
|
|
107
|
+
|
|
108
|
+
The lexer takes the FIRST terminal that matches, by priority, not the longest
|
|
109
|
+
one, so CONSTANT has to match whole words only. Its ``c_`` form ends in a
|
|
110
|
+
negative lookahead for a character that continues a NAME (a letter, digit,
|
|
111
|
+
underscore or combining mark): ``c_new_york`` is then declined by CONSTANT and
|
|
112
|
+
read whole as a NAME, where before CONSTANT cut it at ``c_new`` and the
|
|
113
|
+
remaining ``_york`` could not be read (nine dialects refused the word; only the
|
|
114
|
+
modal dialect's Earley fallback, which weighs every terminal, read it). The
|
|
115
|
+
lookahead names the whole continuation class and not just the underscore,
|
|
116
|
+
because the engine would otherwise back off to ``c_ne`` and match that.
|
|
117
|
+
|
|
118
|
+
WHY THE GENERATED PATTERNS ARE SMALL: LOOKAHEAD, NOT A SECOND EXPLICIT LIST
|
|
119
|
+
-----------------------------------------------------------------------------
|
|
120
|
+
The five terminal patterns below used to be built by literally splicing the
|
|
121
|
+
explicit ``upper``/``lower``/``combining`` class bodies (see ``_classes()``)
|
|
122
|
+
into each terminal's own pattern text — several times each, since a
|
|
123
|
+
terminal's continuation class ("more letters/digits/underscores/marks") had
|
|
124
|
+
to be written out in full at every position it appeared. NAME alone spelled
|
|
125
|
+
that continuation class out three times inside its own pattern (once for
|
|
126
|
+
each of the two ``[cont]*`` runs in its alpha-leading form, once more in its
|
|
127
|
+
digit-leading form) plus the combined upper+lower class a fourth time for
|
|
128
|
+
its "one more letter" requirement; at over 100KB, that ONE terminal was the
|
|
129
|
+
bulk of a combined ~190KB ``terminal_block(include_sort=True)``, and
|
|
130
|
+
``re``/Lark compiling text that size is where the measured slowdown (a bare
|
|
131
|
+
``MSFLParser()`` going from single-digit milliseconds to ~200ms once the
|
|
132
|
+
identifier terminals widened) actually goes — NOT the codepoint scan in
|
|
133
|
+
``_classes()``, which stays well under a tenth of a second and was already
|
|
134
|
+
the one thing this module paid exactly once per process, cached, before and
|
|
135
|
+
after this change; and NOT parse throughput, which is unaffected either way
|
|
136
|
+
(the regex engine still walks the input once per character regardless of
|
|
137
|
+
how its pattern text is spelled).
|
|
138
|
+
|
|
139
|
+
The fix is not a tighter enumeration — ``lower`` is already a minimal
|
|
140
|
+
run-length encoding of a genuinely scattered set (every alphabetic script's
|
|
141
|
+
lowercase block is its own disjoint ``\\uXXXX-\\uYYYY`` span; there are a
|
|
142
|
+
lot of scripts) and cannot get meaningfully smaller as a literal list. The
|
|
143
|
+
fix is to stop writing that list out over and over, using something
|
|
144
|
+
Python's ``re`` module already tests cheaply instead of an enumerated class:
|
|
145
|
+
``\\w``. CPython defines ``\\w`` (in Unicode mode, the only mode this module
|
|
146
|
+
or Lark's parser ever runs in) as matching a codepoint exactly when
|
|
147
|
+
``str.isalnum(c)`` is true or ``c == '_'``, and ``str.isalnum()`` is in turn
|
|
148
|
+
``isalpha() or isdecimal() or isdigit() or isnumeric()``. So
|
|
149
|
+
``[^\\W\\d_]`` — a word character, with decimal digits and underscore
|
|
150
|
+
carved back out via the leading ``\\d``/``_`` in the negated class — is
|
|
151
|
+
``isalpha()`` PLUS one small extra sliver: codepoints that are
|
|
152
|
+
``isdigit()`` or ``isnumeric()`` without being ``isdecimal()`` (already
|
|
153
|
+
excluded by ``\\d``) or ``isalpha()`` — superscript/subscript digits
|
|
154
|
+
(``²``), Roman numerals (``Ⅻ``), vulgar fractions (``½``), circled/
|
|
155
|
+
parenthesized number forms, and similar. That sliver (below, DELTA) is what
|
|
156
|
+
has to be excluded for ``[^\\W\\d_]`` to mean exactly "is a letter"; scanned
|
|
157
|
+
at 0x0000-0x2FFFF it comes to 80 disjoint spans, a little over a kilobyte of
|
|
158
|
+
pattern text — two orders of magnitude smaller than the 12KB-20KB classes it
|
|
159
|
+
stands in for, because it only has to list the exceptions \\w tacks on
|
|
160
|
+
beside "is a letter", not the tens of thousands of letters themselves.
|
|
161
|
+
|
|
162
|
+
So LETTER — "any letter this module recognises", i.e. the exact union of
|
|
163
|
+
:func:`uppercase_class` and :func:`lowercase_class` a continuation position
|
|
164
|
+
used to get by splicing both bodies in directly — becomes
|
|
165
|
+
``(?:[UPPER_NONALPHA]|(?!GREEK)(?!DELTA)(?!CEILING)[^\\W\\d_])``: small,
|
|
166
|
+
FIXED-size pieces, used however many times a terminal's shape needs a
|
|
167
|
+
letter, instead of the same multi-kilobyte enumeration copy-pasted that
|
|
168
|
+
many times. Two alternatives, not one, because ``\\w`` alone cannot stand
|
|
169
|
+
in for "is a letter, either case": ``upper`` (``str.isupper()``) contains a
|
|
170
|
+
120-codepoint sliver — Roman numerals (U+2160-U+216F, category Nl) and
|
|
171
|
+
circled/squared Latin capitals (e.g. U+24B6, category So) — that carries
|
|
172
|
+
no case-STYLE distinction Python calls "alphabetic" at all
|
|
173
|
+
(``str.isalpha()`` is false for every one of them), so ``\\w``-based DELTA
|
|
174
|
+
exclusion or plain non-membership rules them out of the second alternative
|
|
175
|
+
regardless of any lookahead tweak; some are not even ``\\w`` members to
|
|
176
|
+
begin with (So-category symbols are not ``isalnum()``), so no negative
|
|
177
|
+
lookahead over ``\\w`` could ever admit them — a lookahead can only narrow
|
|
178
|
+
what a pattern already matches, never widen it. UPPER_NONALPHA
|
|
179
|
+
(:data:`_Classes.upper_nonalpha`, ``upper`` intersected with "not
|
|
180
|
+
isalpha()") explicitly lists that 120-codepoint sliver instead (five
|
|
181
|
+
contiguous spans, closer to DELTA's size than to UPPER's). Every other
|
|
182
|
+
letter — every codepoint the union actually shares with plain
|
|
183
|
+
``isalpha()`` — still goes through the cheap ``\\w``-based path, with a
|
|
184
|
+
CEILING lookahead alongside GREEK/DELTA so the lookahead atom, which has
|
|
185
|
+
no length limit of its own the way an enumerated class does, cannot admit
|
|
186
|
+
anything :func:`_class_body` did not itself scan up to
|
|
187
|
+
:data:`_MAX_CODEPOINT`. It is a regex ATOM, not a character-class body: it
|
|
188
|
+
opens with a group and lookahead assertions, so — unlike
|
|
189
|
+
:func:`uppercase_class`, :func:`lowercase_class`, and :func:`combining_class`
|
|
190
|
+
below, which still return plain class-body text — it cannot be spliced
|
|
191
|
+
inside a caller's own ``[...]``. Nothing outside this module needs to
|
|
192
|
+
(``dialect_repair.py`` is the only outside consumer of the raw class
|
|
193
|
+
bodies, and it always brackets them itself, so those three functions are
|
|
194
|
+
untouched, same computation and same output as before this section
|
|
195
|
+
existed). UPPER stays fully enumerated regardless: ``re`` has no built-in
|
|
196
|
+
uppercase test the way ``\\w`` doubles as a letter test, so there is nothing
|
|
197
|
+
to invert it out of. And the term-valued ("lowerish") letter — everywhere
|
|
198
|
+
``lowercase_class()``'s ~12KB body used to be spliced in directly — becomes
|
|
199
|
+
LETTER with UPPER's already-computed text reused once more as a negative
|
|
200
|
+
lookahead in front of it, rather than a second, separately-enumerated
|
|
201
|
+
"letter minus upper" class (UPPER's exclusion also correctly screens out
|
|
202
|
+
LETTER's own UPPER_NONALPHA alternative, since that alternative is a
|
|
203
|
+
subset of UPPER).
|
|
204
|
+
|
|
205
|
+
Only the functions that build a whole terminal's regex (``predicate_pattern``
|
|
206
|
+
and its siblings, and ``terminal_block``) were rewritten on these smaller
|
|
207
|
+
atoms. What a caller of those gets back is a differently-spelled pattern for
|
|
208
|
+
the exact same set of strings, never a different one.
|
|
209
|
+
|
|
210
|
+
WHAT SECURES THE EQUIVALENCE
|
|
211
|
+
--------------------------------
|
|
212
|
+
DELTA and UPPER_NONALPHA are computed the same way UPPER/LOWER/COMBINING
|
|
213
|
+
already were — one more call each to :func:`_class_body`, scanning
|
|
214
|
+
0x0000-0x2FFFF once more inside the same cached, once-per-process
|
|
215
|
+
:func:`_classes` — so neither can ever drift from whatever Unicode version
|
|
216
|
+
the running interpreter actually implements, the same guarantee the rest
|
|
217
|
+
of this module already gives; neither is typed in by hand and neither can
|
|
218
|
+
silently go stale across a Python upgrade. But that construction is an
|
|
219
|
+
assertion, not a proof of the claim above (that
|
|
220
|
+
``(?:[UPPER_NONALPHA]|(?!GREEK)(?!DELTA)(?!CEILING)[^\\W\\d_])`` matches
|
|
221
|
+
precisely "is a letter (either case, including the caseless-but-cased
|
|
222
|
+
Roman-numeral/circled-symbol sliver), is not excluded, and is not beyond
|
|
223
|
+
``_MAX_CODEPOINT``"), so ``tests/test_identifiers_equivalence.py`` checks
|
|
224
|
+
it EXHAUSTIVELY rather than on a sample: for every one of the 0x30000
|
|
225
|
+
codepoints from 0x0000 to 0x2FFFF, it confirms the new letter atom matches
|
|
226
|
+
exactly where ``str.isalpha() or str.isupper()`` holds and the codepoint is
|
|
227
|
+
not one of :data:`_EXCLUDED_RANGES`, and that the new term-valued
|
|
228
|
+
("lowerish") atom matches exactly where
|
|
229
|
+
``str.isalpha() and not str.isupper()`` holds under the same exclusion —
|
|
230
|
+
i.e. exactly the sets the old, fully-enumerated ``lower``/``upper`` classes
|
|
231
|
+
contained (in ``lower``'s case exactly; in ``upper``'s case, the LETTER
|
|
232
|
+
atom matches the SAME set ``upper`` does, just split across the two
|
|
233
|
+
alternatives above), codepoint for codepoint. A second test class checks
|
|
234
|
+
codepoints ABOVE ``_MAX_CODEPOINT`` are rejected, since the exhaustive scan
|
|
235
|
+
by construction cannot exercise the CEILING lookahead itself.
|
|
236
|
+
|
|
237
|
+
WHY A CONSTANT MAY BE QUOTED, AND WHEN ITS NAME IS BARE
|
|
238
|
+
-----------------------------------------------------------
|
|
239
|
+
The shape of a bare word decides what it is: ``k2`` is a variable, ``Alice``
|
|
240
|
+
(third-order dialect) a predicate term, ``1`` a number, ``G-910`` no term at
|
|
241
|
+
all. So a constant with such a name had no text, and ``Constant("k2")``
|
|
242
|
+
printed as ``k2``, which reads back as another node. QUOTED_NAME is the way to
|
|
243
|
+
write any constant: ``'k2'``, ``'Alice'``, ``'G-910'``, ``'John Doe'``. Between
|
|
244
|
+
the quotes stand one or more characters, each either an ordinary character
|
|
245
|
+
(anything but a quote, a backslash, a control character, U+0085, U+2028,
|
|
246
|
+
U+2029 or a surrogate) or one of the two escapes ``\\'`` and ``\\\\``. No other
|
|
247
|
+
escape exists, and the empty name has no quoted form. The terminal begins with
|
|
248
|
+
a character no other terminal can begin with, so it competes with none of them
|
|
249
|
+
and needs no priority. It is accepted exactly where a constant stands as a
|
|
250
|
+
term: bare in the unsorted dialects, and only as ``'k2':Mountain`` in the
|
|
251
|
+
sorted ones (which have no bare constant either). Names of functions,
|
|
252
|
+
predicates, variables, sorts, the subscript of a modal operator (``K_a``) and
|
|
253
|
+
nominals have no quoted form.
|
|
254
|
+
|
|
255
|
+
:func:`is_bare_constant` says when the bare text of a name reads back as that
|
|
256
|
+
very constant, and it follows the LEXER, not "some dialect happens to accept
|
|
257
|
+
it": the text has to be one whole CONSTANT token or one whole NAME token (with
|
|
258
|
+
CONSTANT's whole-word rule above, this is a full match of either pattern).
|
|
259
|
+
:func:`constant_text` is the text that reads back as ``Constant(name)``: the
|
|
260
|
+
bare name when it can stand bare, else the name in quotes, with ``'`` written
|
|
261
|
+
``\\'`` and ``\\`` written ``\\\\``. A name that cannot be written at all (not a
|
|
262
|
+
string, empty, or holding a character the quoted form excludes) is refused by
|
|
263
|
+
name rather than printed as text that reads as something else.
|
|
264
|
+
|
|
265
|
+
WHAT THIS MODULE DOES NOT TOUCH
|
|
266
|
+
-----------------------------------
|
|
267
|
+
NUMBER (``[0-9]+(\\.[0-9]+)?``), FORALL, EXISTS, and LAMBDA stay exactly as
|
|
268
|
+
declared in ``fol/grammars/terminals.lark`` — none of them classify a
|
|
269
|
+
letter, so none of them need widening, and none of them are generated here.
|
|
270
|
+
``fol/sanitize.py``, the layer that maps AST names down to strictly-ASCII
|
|
271
|
+
tokens for TPTP/Prover9/SMT-LIB/Isabelle/CASL export, is untouched on
|
|
272
|
+
purpose: it exists to make names safe for ASCII-only target formats, not to
|
|
273
|
+
make them parseable by this (now Unicode-wide) grammar, and widening it
|
|
274
|
+
would leak raw non-ASCII text into export formats that cannot represent it.
|
|
275
|
+
|
|
276
|
+
PUBLIC SURFACE
|
|
277
|
+
------------------
|
|
278
|
+
:func:`uppercase_class`, :func:`lowercase_class`, :func:`combining_class`
|
|
279
|
+
return the three raw regex character-CLASS BODIES (no enclosing ``[``/``]``)
|
|
280
|
+
this module is built from, for reuse by anything that needs to recognise the
|
|
281
|
+
same alphabet outside the grammar (``fol/dialect_repair.py``'s legality
|
|
282
|
+
check, and tests) by splicing them inside its own ``[...]``. They are
|
|
283
|
+
unaffected by the LOOKAHEAD section above: same computation, same returned
|
|
284
|
+
text, before and after. :func:`predicate_pattern`, :func:`name_pattern`,
|
|
285
|
+
:func:`constant_pattern`, :func:`variable_pattern`, :func:`sort_pattern` and
|
|
286
|
+
:func:`quoted_name_pattern`
|
|
287
|
+
return the full terminal regex (again without a Lark terminal name or
|
|
288
|
+
priority — just the pattern text between the ``/.../``) for each widened
|
|
289
|
+
terminal, now built from the small internal lookahead atoms rather than
|
|
290
|
+
repeated copies of the class bodies. :func:`is_variable_name`,
|
|
291
|
+
:func:`is_bare_constant` and :func:`constant_text` are the questions the
|
|
292
|
+
terminal shapes answer about one name (is it a variable; does its bare text
|
|
293
|
+
read back as that constant; what text reads back as it) and are exported from
|
|
294
|
+
the package. :func:`terminal_block` renders the
|
|
295
|
+
complete, ready-to-splice Lark terminal declarations (name, priority, and
|
|
296
|
+
pattern together) that ``_fol_nodes.build_grammar`` inserts into every
|
|
297
|
+
mode's grammar text. :data:`HUMAN_READABLE_PATTERNS` maps each widened
|
|
298
|
+
terminal's name to a short English description of its shape, for
|
|
299
|
+
``fol/naming.py`` to show in a NamingError instead of the generated regex
|
|
300
|
+
text.
|
|
301
|
+
|
|
302
|
+
The same module is the one place a MINTED name gets its shape, because the
|
|
303
|
+
shapes are the terminals' business: :func:`fresh_variables` (a batch of
|
|
304
|
+
``letter`` + digits), :func:`fresh_variable_like` (one alpha-renamed binder,
|
|
305
|
+
keeping the old letter), :func:`fresh_like` (a lambda parameter, which keeps its
|
|
306
|
+
kind: variable, NAME or PREDICATE) and :func:`variable_names` (what a minted
|
|
307
|
+
name has to avoid). Anything that prints a formula the kit cannot read back has
|
|
308
|
+
minted a name some other way.
|
|
309
|
+
"""
|
|
310
|
+
|
|
311
|
+
import dataclasses
|
|
312
|
+
import re
|
|
313
|
+
import unicodedata
|
|
314
|
+
from functools import lru_cache
|
|
315
|
+
from typing import Callable, NamedTuple, Optional
|
|
316
|
+
|
|
317
|
+
__all__ = [
|
|
318
|
+
"uppercase_class", "lowercase_class", "combining_class",
|
|
319
|
+
"predicate_pattern", "name_pattern", "constant_pattern",
|
|
320
|
+
"variable_pattern", "sort_pattern", "quoted_name_pattern",
|
|
321
|
+
"terminal_block", "HUMAN_READABLE_PATTERNS",
|
|
322
|
+
"is_variable_name", "is_bare_constant", "constant_text",
|
|
323
|
+
"fresh_variables", "fresh_variable_like", "fresh_like", "variable_names",
|
|
324
|
+
"symbol_names",
|
|
325
|
+
]
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
# ---------------------------------------------------------------------------
|
|
329
|
+
# Fresh names the kit's own parser reads back
|
|
330
|
+
#
|
|
331
|
+
# Every generator that mints a bound name (alpha-renaming, witnesses for a
|
|
332
|
+
# counting quantifier, a relativisation guard, a frame axiom, ...) has to mint a
|
|
333
|
+
# name of the SHAPE the position needs, or the text it prints is text the kit
|
|
334
|
+
# cannot read back: ``∀y_0 R(y, y_0)`` was printed by capture-avoiding
|
|
335
|
+
# substitution and rejected by ``api.parse_any``. These helpers are the one place
|
|
336
|
+
# such names are made.
|
|
337
|
+
#
|
|
338
|
+
# The shapes, from the terminals above:
|
|
339
|
+
# * an object variable: ONE term-valued letter, then ASCII digits only
|
|
340
|
+
# (``y``, ``y0``, ``y12``) -- no underscore, no prefix;
|
|
341
|
+
# * a NAME (function symbol, lambda parameter): two or more letters, and
|
|
342
|
+
# underscores and digits are fine after the first (``foo_0``);
|
|
343
|
+
# * a PREDICATE (second-order variable, lambda parameter): an uppercase-led
|
|
344
|
+
# run that likewise takes ``_`` and digits (``P_0``).
|
|
345
|
+
# ---------------------------------------------------------------------------
|
|
346
|
+
|
|
347
|
+
#: ``[a-z][0-9]*`` -- the ASCII core of VARIABLE. Every string it accepts the full
|
|
348
|
+
#: VARIABLE terminal accepts too, and it needs no scan of the Unicode tables, so
|
|
349
|
+
#: the common case (``x``, ``y1``) never pays the one-off start-up cost of
|
|
350
|
+
#: :func:`variable_pattern`.
|
|
351
|
+
_ASCII_VARIABLE = re.compile(r"[a-z][0-9]*")
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
#: ASCII core of a BARE constant: the strings of ASCII characters that the CONSTANT
|
|
355
|
+
#: terminal (its ``c_`` form) or the NAME terminal accepts as one whole token. Written
|
|
356
|
+
#: out by hand so that the common case never pays the one-off scan of the Unicode
|
|
357
|
+
#: tables; ``tests/test_identifiers_equivalence.py`` pins it against the generated patterns.
|
|
358
|
+
#:
|
|
359
|
+
#: * ``[a-z][0-9_]*[a-zA-Z][a-zA-Z0-9_]*`` -- NAME led by a letter: the first letter
|
|
360
|
+
#: (lower case), then, up to the SECOND letter, only digits and underscores, then the
|
|
361
|
+
#: rest. That is "at least two letters" without a backtracking run.
|
|
362
|
+
#: * ``[0-9]+[a-zA-Z][a-zA-Z0-9_]*`` -- NAME led by digits.
|
|
363
|
+
#: * ``c_[a-zA-Z0-9]+`` -- the ``c_`` form of CONSTANT (its tail has no underscore).
|
|
364
|
+
_ASCII_BARE_CONSTANT = re.compile(
|
|
365
|
+
r"[a-z][0-9_]*[a-zA-Z][a-zA-Z0-9_]*"
|
|
366
|
+
r"|[0-9]+[a-zA-Z][a-zA-Z0-9_]*"
|
|
367
|
+
r"|c_[a-zA-Z0-9]+")
|
|
368
|
+
|
|
369
|
+
#: The characters a constant's name cannot hold in ANY spelling, bare or quoted: the
|
|
370
|
+
#: control characters (a line break would end the statement of the text that holds the
|
|
371
|
+
#: formula), DEL, NEL, the two Unicode line and paragraph separators, and surrogates
|
|
372
|
+
#: (which no text encoding can carry). The body of a character class, shared by the
|
|
373
|
+
#: QUOTED_NAME terminal and by the printer's refusal, so the two cannot drift apart.
|
|
374
|
+
_UNSPELLABLE_CLASS = r"\x00-\x1f\x7f\x85\u2028\u2029\ud800-\udfff"
|
|
375
|
+
_UNSPELLABLE = re.compile(f"[{_UNSPELLABLE_CLASS}]")
|
|
376
|
+
|
|
377
|
+
#: A backslash in front of a quote or of a backslash: the only two escapes a quoted
|
|
378
|
+
#: name has. ``_CHARACTER_TO_ESCAPE`` finds the characters that need one.
|
|
379
|
+
_ESCAPED_CHARACTER = re.compile(r"\\(['\\])")
|
|
380
|
+
_CHARACTER_TO_ESCAPE = re.compile(r"(['\\])")
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
@lru_cache(maxsize=None)
|
|
384
|
+
def _compiled(which: str):
|
|
385
|
+
"""The compiled terminal pattern called ``which`` (cached per process)."""
|
|
386
|
+
builder = {"variable": variable_pattern, "name": name_pattern,
|
|
387
|
+
"predicate": predicate_pattern, "constant": constant_pattern}[which]
|
|
388
|
+
return re.compile(builder())
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
def is_variable_name(text) -> bool:
|
|
392
|
+
"""Whether the VARIABLE terminal accepts ``text`` as one whole token.
|
|
393
|
+
|
|
394
|
+
One term-valued letter followed by ASCII digits and nothing else: ``x``,
|
|
395
|
+
``y12``, ``é``, ``北``. So ``x`` as a term is always the variable, and a
|
|
396
|
+
constant of that name has to be written in quotes (``'x'``). ``text`` that is
|
|
397
|
+
not a string is no variable name.
|
|
398
|
+
"""
|
|
399
|
+
if not isinstance(text, str):
|
|
400
|
+
return False
|
|
401
|
+
return bool(_ASCII_VARIABLE.fullmatch(text)) or bool(
|
|
402
|
+
_compiled("variable").fullmatch(text))
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
#: How many names the two decisions below remember. The printer asks about every constant
|
|
406
|
+
#: of every formula it writes, and the constants of one problem are few and recur, so the
|
|
407
|
+
#: answer is looked up and not worked out again. A name is a string, so it is its own key.
|
|
408
|
+
_MEMORY = 8192
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
@lru_cache(maxsize=_MEMORY)
|
|
412
|
+
def _is_bare_constant(name: str) -> bool:
|
|
413
|
+
""":func:`is_bare_constant` for a string (no type check)."""
|
|
414
|
+
if name.isascii():
|
|
415
|
+
return _ASCII_BARE_CONSTANT.fullmatch(name) is not None
|
|
416
|
+
return (_compiled("constant").fullmatch(name) is not None
|
|
417
|
+
or _compiled("name").fullmatch(name) is not None)
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def is_bare_constant(name) -> bool:
|
|
421
|
+
"""Whether the bare text ``name`` reads back as ``Constant(name)`` as a term.
|
|
422
|
+
|
|
423
|
+
"Reads back" is what the LEXER does: it takes the first terminal that
|
|
424
|
+
matches, by priority, not the longest, so the text has to be ONE whole
|
|
425
|
+
CONSTANT token or ONE whole NAME token. CONSTANT matches whole words only (a
|
|
426
|
+
``c_`` word that continues, ``c_new_york``, is not CONSTANT's but NAME's), so
|
|
427
|
+
this is a full match of either pattern. Hence ``socrates``, ``c_k2``,
|
|
428
|
+
``c_new_york``, ``θ``, ``2008SummerOlympics`` and ``świątek`` are bare, and
|
|
429
|
+
``a`` and ``k2`` (variables), ``Alice`` (a predicate), ``1`` and ``-3``
|
|
430
|
+
(numbers), ``G-910``, ``C++``, ``a b``, ``_sk0``, ``λ`` and ``x_1`` are not.
|
|
431
|
+
|
|
432
|
+
A name that is not a string, or is empty, is not bare.
|
|
433
|
+
"""
|
|
434
|
+
return isinstance(name, str) and _is_bare_constant(name)
|
|
435
|
+
|
|
436
|
+
|
|
437
|
+
def constant_text(name) -> str:
|
|
438
|
+
"""The text that reads back as ``Constant(name)`` in term position.
|
|
439
|
+
|
|
440
|
+
The bare ``name`` when :func:`is_bare_constant` holds, else the name in single
|
|
441
|
+
quotes, with a quote written ``\\'`` and a backslash written ``\\\\``:
|
|
442
|
+
``socrates`` stays ``socrates``, ``k2`` becomes ``'k2'``, ``it's`` becomes
|
|
443
|
+
``'it\\'s'``. A sorted constant takes the same text of its name
|
|
444
|
+
(``socrates:Human``, ``'k2':Mountain``).
|
|
445
|
+
|
|
446
|
+
Raises:
|
|
447
|
+
TypeError: ``name`` is not a string.
|
|
448
|
+
ValueError: ``name`` is empty, or holds a control character (U+0000 to
|
|
449
|
+
U+001F, U+007F), U+0085, U+2028, U+2029 or a surrogate, which no
|
|
450
|
+
spelling can carry.
|
|
451
|
+
"""
|
|
452
|
+
if not isinstance(name, str):
|
|
453
|
+
raise TypeError(
|
|
454
|
+
f"constant_text: the name of a constant must be a string, got "
|
|
455
|
+
f"{name!r} (a {type(name).__name__}). A constant without a string name "
|
|
456
|
+
f"has no text; give it one.")
|
|
457
|
+
return _constant_text(name)
|
|
458
|
+
|
|
459
|
+
|
|
460
|
+
@lru_cache(maxsize=_MEMORY)
|
|
461
|
+
def _constant_text(name: str) -> str:
|
|
462
|
+
""":func:`constant_text` for a string (no type check). An exception is not remembered."""
|
|
463
|
+
if _is_bare_constant(name):
|
|
464
|
+
return name
|
|
465
|
+
if not name:
|
|
466
|
+
raise ValueError(
|
|
467
|
+
f"constant_text: the constant {name!r} has an empty name, so it has no text: "
|
|
468
|
+
f"no bare word is empty, and two quotes in a row are no constant. Give it a "
|
|
469
|
+
f"name of at least one character.")
|
|
470
|
+
excluded = _UNSPELLABLE.search(name)
|
|
471
|
+
if excluded is not None:
|
|
472
|
+
raise ValueError(
|
|
473
|
+
f"constant_text: the constant named {name!r} has no text: it holds "
|
|
474
|
+
f"{excluded.group()!r} (U+{ord(excluded.group()):04X}), a character that "
|
|
475
|
+
f"neither the bare nor the quoted form can carry (control characters, "
|
|
476
|
+
f"U+007F, U+0085, U+2028, U+2029 and surrogates are excluded). Rename "
|
|
477
|
+
f"the constant, or drop that character from its name.")
|
|
478
|
+
return "'" + _CHARACTER_TO_ESCAPE.sub(r"\\\1", name) + "'"
|
|
479
|
+
|
|
480
|
+
|
|
481
|
+
def _unquote_constant(token_text: str) -> str:
|
|
482
|
+
"""The name that a QUOTED_NAME token spells: the inverse of :func:`constant_text`.
|
|
483
|
+
|
|
484
|
+
``token_text`` is the whole token, quotes included. Inside, a backslash always
|
|
485
|
+
stands in front of a quote or a backslash (the terminal admits no other escape),
|
|
486
|
+
so one left-to-right pass undoes the escapes.
|
|
487
|
+
"""
|
|
488
|
+
return _ESCAPED_CHARACTER.sub(r"\1", token_text[1:-1])
|
|
489
|
+
|
|
490
|
+
|
|
491
|
+
def variable_names(*nodes) -> frozenset:
|
|
492
|
+
"""Every variable name occurring in ``nodes``, bound or free.
|
|
493
|
+
|
|
494
|
+
What a minted name has to avoid: see :func:`fresh_variables`. Three kinds of
|
|
495
|
+
occurrence count, because a fresh name that equals any of them changes what
|
|
496
|
+
the formula says:
|
|
497
|
+
|
|
498
|
+
* an object ``Variable`` and every quantifier / counting binder;
|
|
499
|
+
* a lambda-bound ``LambdaVar`` (the text ``λy. … y …`` reads an object
|
|
500
|
+
variable ``y`` under it as the lambda's own parameter, so the two kinds
|
|
501
|
+
cannot share a name);
|
|
502
|
+
* a name in an IF-logic slash set (``∃y/{z} …``): it is a plain string, not
|
|
503
|
+
a node, yet it refers to an enclosing variable.
|
|
504
|
+
"""
|
|
505
|
+
names = set()
|
|
506
|
+
for node in nodes:
|
|
507
|
+
for inner in node.walk():
|
|
508
|
+
name = getattr(inner, "name", None)
|
|
509
|
+
if name is not None and type(inner).__name__ in ("Variable", "LambdaVar"):
|
|
510
|
+
names.add(name)
|
|
511
|
+
# a quantifier's own binder is a Variable in .variable
|
|
512
|
+
binder = getattr(inner, "variable", None)
|
|
513
|
+
if binder is not None and getattr(binder, "name", None):
|
|
514
|
+
names.add(binder.name)
|
|
515
|
+
names.update(getattr(inner, "slashed", ()) or ())
|
|
516
|
+
return frozenset(names)
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
def symbol_names(*nodes, fold: Optional[Callable[[str], str]] = None) -> frozenset:
|
|
520
|
+
"""Every name that ``nodes`` carry, of every kind: what a MINTED name has to avoid.
|
|
521
|
+
|
|
522
|
+
:func:`variable_names` is the avoid set for a new bound variable inside one
|
|
523
|
+
formula of this kit, where a variable can only meet another variable. A name
|
|
524
|
+
that is minted for a TARGET -- a Skolem constant, a tableau parameter, a
|
|
525
|
+
tracking literal of a solver, a witness written into SMT-LIB or Prover9
|
|
526
|
+
text, a node of a description-logic tableau -- can meet any symbol there:
|
|
527
|
+
a constant ``_sk0``, a proposition ``goal``, a predicate ``x0``, a sort
|
|
528
|
+
``x0``. So this returns every string a node of the formulas holds: the name
|
|
529
|
+
of a predicate, function, constant, sorted constant, variable, nominal or
|
|
530
|
+
agent, the sort of a sorted node, the names in a slash set. It is an
|
|
531
|
+
over-approximation on purpose (the glyph of a quantifier is a string a node
|
|
532
|
+
holds too): for an avoid set a name too many costs nothing, and a node class
|
|
533
|
+
added later is covered without being listed here.
|
|
534
|
+
|
|
535
|
+
Pass every formula of the problem -- the premises, the conclusion and the
|
|
536
|
+
background sentences that are added -- or the name is fresh for one formula
|
|
537
|
+
and taken in the next.
|
|
538
|
+
|
|
539
|
+
``fold`` maps each name to the form the target compares names in:
|
|
540
|
+
``str.casefold`` for a target that reads ``x0`` and ``X0`` as one word (TPTP
|
|
541
|
+
and Prover9 fold the case of a first letter), nothing for a target with
|
|
542
|
+
case-sensitive names. Mint in the same form: a candidate is fresh when
|
|
543
|
+
``fold(candidate)`` is not in the result.
|
|
544
|
+
"""
|
|
545
|
+
names = set()
|
|
546
|
+
for node in nodes:
|
|
547
|
+
for inner in node.walk():
|
|
548
|
+
values = (getattr(inner, field.name, None) for field in dataclasses.fields(inner)) \
|
|
549
|
+
if dataclasses.is_dataclass(inner) else vars(inner).values()
|
|
550
|
+
for value in values:
|
|
551
|
+
if isinstance(value, str):
|
|
552
|
+
names.add(value)
|
|
553
|
+
elif isinstance(value, (tuple, list, set, frozenset)):
|
|
554
|
+
names.update(item for item in value if isinstance(item, str))
|
|
555
|
+
return frozenset(fold(name) for name in names) if fold is not None else frozenset(names)
|
|
556
|
+
|
|
557
|
+
|
|
558
|
+
def fresh_variables(count: int, *, letter: str = "x", avoid=()) -> tuple:
|
|
559
|
+
"""``count`` variable names that the VARIABLE terminal actually accepts.
|
|
560
|
+
|
|
561
|
+
A translation that mints its own bound variables has to mint names the
|
|
562
|
+
kit's own parser reads back, or its output is a formula the kit cannot
|
|
563
|
+
re-read: ``∃x_1 (…)``, ``∀_hw0 R(_hw0, _hw0)`` and
|
|
564
|
+
``∃_msfol_Human_witness Human(…)`` were all printed by the kit and all
|
|
565
|
+
rejected by :func:`unicode_logic_kit.api.parse_any`, because VARIABLE is one
|
|
566
|
+
term-valued letter followed by ASCII DIGITS only — no underscore, no
|
|
567
|
+
prefix (see :func:`variable_pattern`).
|
|
568
|
+
|
|
569
|
+
So the shape here is ``letter`` + digits (``x0``, ``x1``, …), skipping
|
|
570
|
+
everything in ``avoid`` — pass :func:`variable_names` of whatever the
|
|
571
|
+
result will sit next to. Collisions are only a READABILITY matter for
|
|
572
|
+
bound variables (two quantifiers may bind the same name without either
|
|
573
|
+
capturing the other), but a collision with a FREE variable of the host
|
|
574
|
+
formula would change its meaning, which is what ``avoid`` is for.
|
|
575
|
+
|
|
576
|
+
Raises:
|
|
577
|
+
ValueError: ``letter`` is not a single character the terminal accepts.
|
|
578
|
+
"""
|
|
579
|
+
if not is_variable_name(letter):
|
|
580
|
+
raise ValueError(
|
|
581
|
+
f"fresh_variables: {letter!r} is not a legal variable name on its "
|
|
582
|
+
f"own, so {letter!r} + digits is not one either")
|
|
583
|
+
avoid = set(avoid)
|
|
584
|
+
out: list = []
|
|
585
|
+
index = 0
|
|
586
|
+
while len(out) < count:
|
|
587
|
+
candidate = f"{letter}{index}"
|
|
588
|
+
index += 1
|
|
589
|
+
if candidate not in avoid:
|
|
590
|
+
out.append(candidate)
|
|
591
|
+
return tuple(out)
|
|
592
|
+
|
|
593
|
+
|
|
594
|
+
def _variable_letter(base: str) -> str:
|
|
595
|
+
"""The letter an alpha-renamed variable called ``base`` keeps.
|
|
596
|
+
|
|
597
|
+
``base``'s own first character when the VARIABLE terminal accepts it (so
|
|
598
|
+
``y`` is renamed to ``y0``, not to an unrelated letter), else its lowercase
|
|
599
|
+
form (a Prolog-style ``Y`` becomes ``y0``), else ``x``. The last two cover a
|
|
600
|
+
binder whose name was not minted by this kit -- an imported or hand-built
|
|
601
|
+
``Variable("_tmp")`` -- which still has to be renameable: renaming a binder to
|
|
602
|
+
a fresh LEGAL name is exactly alpha-equivalence, whatever it was called.
|
|
603
|
+
"""
|
|
604
|
+
first = base[:1]
|
|
605
|
+
for candidate in (first, first.lower()):
|
|
606
|
+
if candidate and is_variable_name(candidate):
|
|
607
|
+
return candidate
|
|
608
|
+
return "x"
|
|
609
|
+
|
|
610
|
+
|
|
611
|
+
def fresh_variable_like(base: str, avoid=()) -> str:
|
|
612
|
+
"""One fresh object-variable name for a binder that is called ``base``.
|
|
613
|
+
|
|
614
|
+
The alpha-renaming counterpart of :func:`fresh_variables`: ``letter`` is
|
|
615
|
+
taken from ``base`` (see ``_variable_letter``), the digits count up from 0,
|
|
616
|
+
and every name in ``avoid`` is skipped. The result is always a legal
|
|
617
|
+
VARIABLE, which is why it is the right name for any quantifier, counting or
|
|
618
|
+
cardinality binder -- whatever kind of name ``base`` was.
|
|
619
|
+
|
|
620
|
+
``avoid`` must hold EVERY name that could be confused with the new binder:
|
|
621
|
+
the free variables of whatever gets substituted in, and every name -- bound
|
|
622
|
+
ones too -- inside the scope being renamed, because renaming ``y`` to a name
|
|
623
|
+
that an inner binder already uses captures the occurrences that moved.
|
|
624
|
+
:func:`variable_names` collects exactly that.
|
|
625
|
+
"""
|
|
626
|
+
return fresh_variables(1, letter=_variable_letter(base), avoid=avoid)[0]
|
|
627
|
+
|
|
628
|
+
|
|
629
|
+
def fresh_like(base: str, avoid=()) -> str:
|
|
630
|
+
"""A fresh name of the same terminal kind as ``base``, for a lambda parameter.
|
|
631
|
+
|
|
632
|
+
A lambda parameter may be a VARIABLE (``λx.``), a NAME (``λfoo.``) or a
|
|
633
|
+
PREDICATE (``λP.``), and the body uses it in the matching position -- as an
|
|
634
|
+
argument, or as the head of an application -- so renaming it must keep its
|
|
635
|
+
kind. A VARIABLE is renamed like any variable (``y`` → ``y0``); a NAME or a
|
|
636
|
+
PREDICATE takes a ``_N`` suffix (``foo`` → ``foo_0``, ``P`` → ``P_0``), which
|
|
637
|
+
is legal for both because underscore and digits are continuation characters
|
|
638
|
+
of either terminal. A ``base`` that is none of the three (it was not made by
|
|
639
|
+
the kit's parser) is renamed to a legal variable.
|
|
640
|
+
"""
|
|
641
|
+
avoid = set(avoid)
|
|
642
|
+
if not is_variable_name(base) and (_compiled("name").fullmatch(base)
|
|
643
|
+
or _compiled("predicate").fullmatch(base)):
|
|
644
|
+
index = 0
|
|
645
|
+
while True:
|
|
646
|
+
candidate = f"{base}_{index}"
|
|
647
|
+
index += 1
|
|
648
|
+
if candidate not in avoid:
|
|
649
|
+
return candidate
|
|
650
|
+
return fresh_variable_like(base, avoid)
|
|
651
|
+
|
|
652
|
+
# Single-character operator glyphs that Unicode also classifies as uppercase
|
|
653
|
+
# letters. Each is a registered operator symbol (``_fol_nodes.OPERATORS``), and
|
|
654
|
+
# each satisfies ``str.isupper()`` — which is precisely the test this module
|
|
655
|
+
# uses to decide "this opens a PREDICATE". Before 0.23.2 that made ``ⓄP`` two
|
|
656
|
+
# things at once, ``Obligatory(P)`` and "the predicate named ⓄP", and only the
|
|
657
|
+
# Earley parser's willingness to pick one hid it: asked for every derivation,
|
|
658
|
+
# it reports the node as ambiguous, and a table-driven lexer takes the other
|
|
659
|
+
# reading. Same rule as Greek below, for the same reason — a glyph that is an
|
|
660
|
+
# operator ANYWHERE is not an identifier character ANYWHERE, even in a mode
|
|
661
|
+
# where that operator is not registered.
|
|
662
|
+
#
|
|
663
|
+
# Only SINGLE-character symbols belong here. The multi-character ones (``K_``,
|
|
664
|
+
# ``B_``, ``Say_``, ``Want_``) open with ordinary letters that obviously cannot
|
|
665
|
+
# be carved out, and do not need to be: their terminals are longer than the
|
|
666
|
+
# prefix they share with a name, so longest-match settles it — ``K_alice``
|
|
667
|
+
# lexes as one KNOWS token, with no ambiguous node. Both halves of that are
|
|
668
|
+
# checked in tests/test_operator_glyphs.py against the live registry, so a
|
|
669
|
+
# newly registered letter-like operator fails there rather than silently
|
|
670
|
+
# becoming a name.
|
|
671
|
+
_OPERATOR_GLYPHS = (
|
|
672
|
+
0x24B8, # Ⓒ CIRCLED LATIN CAPITAL LETTER C — Contrast, every classical mode
|
|
673
|
+
0x24BB, # Ⓕ CIRCLED LATIN CAPITAL LETTER F — Eventually (temporal)
|
|
674
|
+
0x24BC, # Ⓖ CIRCLED LATIN CAPITAL LETTER G — Always (temporal)
|
|
675
|
+
0x24C3, # Ⓝ CIRCLED LATIN CAPITAL LETTER N — Next (temporal)
|
|
676
|
+
0x24C4, # Ⓞ CIRCLED LATIN CAPITAL LETTER O — Obligatory (deontic)
|
|
677
|
+
0x24C5, # Ⓟ CIRCLED LATIN CAPITAL LETTER P — Permitted (deontic)
|
|
678
|
+
0x24CA, # Ⓤ CIRCLED LATIN CAPITAL LETTER U — Until (temporal)
|
|
679
|
+
)
|
|
680
|
+
|
|
681
|
+
# Greek and Coptic, Greek Extended, and the OHM SIGN — see the module
|
|
682
|
+
# docstring's "WHY GREEK AND OHM ARE CARVED OUT" section — plus the operator
|
|
683
|
+
# glyphs above. Note what is NOT here: the other circled capitals (Ⓐ, Ⓑ, …)
|
|
684
|
+
# and the Roman numerals stay legal identifier characters, because they are
|
|
685
|
+
# not operators. The carve-out is a list of symbols the grammar already
|
|
686
|
+
# spends, not a swipe at a Unicode block.
|
|
687
|
+
_EXCLUDED_RANGES = (
|
|
688
|
+
((0x0370, 0x03FF), (0x1F00, 0x1FFF), (0x2126, 0x2126))
|
|
689
|
+
+ tuple((cp, cp) for cp in _OPERATOR_GLYPHS)
|
|
690
|
+
)
|
|
691
|
+
|
|
692
|
+
# Plane 0 (BMP) through Plane 2 (CJK Extension B/C/... territory). Every
|
|
693
|
+
# script this kit's test corpora (FOLIO included) actually exercise lives
|
|
694
|
+
# below this, and the scan is a fixed one-time-per-process cost regardless
|
|
695
|
+
# of where the ceiling sits, so there is no pressure to trim it further.
|
|
696
|
+
_MAX_CODEPOINT = 0x2FFFF
|
|
697
|
+
|
|
698
|
+
|
|
699
|
+
def _excluded(codepoint: int) -> bool:
|
|
700
|
+
return any(lo <= codepoint <= hi for lo, hi in _EXCLUDED_RANGES)
|
|
701
|
+
|
|
702
|
+
|
|
703
|
+
def _escape(codepoint: int) -> str:
|
|
704
|
+
return f"\\u{codepoint:04x}" if codepoint <= 0xFFFF else f"\\U{codepoint:08x}"
|
|
705
|
+
|
|
706
|
+
|
|
707
|
+
def _class_body(predicate: Callable[[str], bool]) -> str:
|
|
708
|
+
"""A regex character-class body (no enclosing ``[``/``]``) matching every
|
|
709
|
+
codepoint up to :data:`_MAX_CODEPOINT` for which ``predicate(chr(cp))``
|
|
710
|
+
holds and the codepoint is not one of :data:`_EXCLUDED_RANGES`, run-
|
|
711
|
+
length-encoded into ``\\uXXXX-\\uYYYY`` spans (astral codepoints use the
|
|
712
|
+
8-digit ``\\U........`` form) so the result is a few kilobytes of regex
|
|
713
|
+
text rather than tens of thousands of single-character alternatives."""
|
|
714
|
+
spans = []
|
|
715
|
+
start = None
|
|
716
|
+
for codepoint in range(_MAX_CODEPOINT + 1):
|
|
717
|
+
keep = predicate(chr(codepoint)) and not _excluded(codepoint)
|
|
718
|
+
if keep and start is None:
|
|
719
|
+
start = codepoint
|
|
720
|
+
elif not keep and start is not None:
|
|
721
|
+
spans.append((start, codepoint - 1))
|
|
722
|
+
start = None
|
|
723
|
+
if start is not None:
|
|
724
|
+
spans.append((start, _MAX_CODEPOINT))
|
|
725
|
+
return "".join(
|
|
726
|
+
_escape(lo) if lo == hi else f"{_escape(lo)}-{_escape(hi)}"
|
|
727
|
+
for lo, hi in spans
|
|
728
|
+
)
|
|
729
|
+
|
|
730
|
+
|
|
731
|
+
class _Classes(NamedTuple):
|
|
732
|
+
upper: str
|
|
733
|
+
lower: str
|
|
734
|
+
combining: str
|
|
735
|
+
delta: str
|
|
736
|
+
upper_nonalpha: str
|
|
737
|
+
|
|
738
|
+
|
|
739
|
+
@lru_cache(maxsize=1)
|
|
740
|
+
def _classes() -> _Classes:
|
|
741
|
+
"""Compute the four character classes this module is built from, once
|
|
742
|
+
per process. Deferred behind ``lru_cache`` rather than run at import
|
|
743
|
+
time: importing this module (or anything that imports it, which given
|
|
744
|
+
``_fol_nodes.py``'s import means most of the package) must stay cheap
|
|
745
|
+
even for a caller who never actually builds a parser; the codepoint
|
|
746
|
+
scan is paid only when a terminal pattern is first asked for — in
|
|
747
|
+
practice, the first time a grammar for some mode is actually built —
|
|
748
|
+
and never again in that process.
|
|
749
|
+
|
|
750
|
+
``upper`` — codepoints with ``str.isupper() == True``: the PREDICATE-
|
|
751
|
+
signalling class (see the module docstring's "WHY THE FIRST
|
|
752
|
+
CHARACTER DECIDES" section).
|
|
753
|
+
``lower`` — every other alphabetic codepoint (``str.isalpha()`` true,
|
|
754
|
+
``str.isupper()`` false): both ordinary lowercase letters and every
|
|
755
|
+
letter from a script with no case distinction at all, which is
|
|
756
|
+
exactly the "term-valued" class that section describes. Returned
|
|
757
|
+
as-is by :func:`lowercase_class` for outside callers; the terminal
|
|
758
|
+
patterns below no longer splice this in directly (see the module
|
|
759
|
+
docstring's LOOKAHEAD section) but it is still computed and
|
|
760
|
+
returned unchanged, since ``dialect_repair.py`` still needs it.
|
|
761
|
+
``combining`` — codepoints in Unicode general categories Mn (nonspacing
|
|
762
|
+
mark) and Mc (spacing combining mark): NFD-decomposed accents such
|
|
763
|
+
as U+0301 COMBINING ACUTE ACCENT, needed so that ``świątek``
|
|
764
|
+
survives NFD normalisation (``ś`` → ``s`` + U+0301) and still lexes
|
|
765
|
+
as one identifier rather than two tokens plus a stray mark. Not
|
|
766
|
+
part of ``\\w`` (marks are not alphanumeric), so — unlike ``lower``
|
|
767
|
+
— there is no cheaper way to express this one; it stays a literal
|
|
768
|
+
enumerated class either way.
|
|
769
|
+
``delta`` — codepoints ``\\w`` matches that ``[^\\W\\d_]`` alone would
|
|
770
|
+
wrongly admit as letters: ``isdigit()`` or ``isnumeric()`` without
|
|
771
|
+
being ``isdecimal()`` (already excluded by ``\\d``) or
|
|
772
|
+
``isalpha()``. Not returned by any public function — nothing
|
|
773
|
+
outside this module needs the raw sliver, only the negative
|
|
774
|
+
lookahead built from it (see :func:`_letter_atom`).
|
|
775
|
+
``upper_nonalpha`` — codepoints with ``str.isupper() == True`` but
|
|
776
|
+
``str.isalpha() == False``: e.g. the Roman numeral block
|
|
777
|
+
U+2160-U+216F (category Nl, ``isupper()`` true, ``isalpha()``
|
|
778
|
+
false) and circled/squared Latin capitals such as U+24B6 (category
|
|
779
|
+
So). ``upper`` already contains these (it is built from
|
|
780
|
+
``str.isupper()`` alone, with no ``isalpha()`` condition), so a
|
|
781
|
+
continuation position that used to splice ``upper`` in directly
|
|
782
|
+
(PREDICATE/SORT/NAME's own continuation, before this module's
|
|
783
|
+
lookahead rewrite) accepted them; a LETTER atom built purely from
|
|
784
|
+
``\\w`` cannot reach them the same way — some are not even ``\\w``
|
|
785
|
+
members (So-category symbols are not ``isalnum()``), and the rest
|
|
786
|
+
(the Nl-category Roman numerals) are exactly what ``delta`` above
|
|
787
|
+
excludes. So they need a small explicit class of their own rather
|
|
788
|
+
than a lookahead tweak (see :func:`_letter_atom`); it is tiny (five
|
|
789
|
+
contiguous spans, 120 codepoints total) because it is exactly
|
|
790
|
+
``upper`` minus ``lower``'s complement restricted to ``isupper()``
|
|
791
|
+
already-non-alphabetic codepoints, not a general enumeration.
|
|
792
|
+
"""
|
|
793
|
+
upper = _class_body(str.isupper)
|
|
794
|
+
lower = _class_body(lambda ch: ch.isalpha() and not ch.isupper())
|
|
795
|
+
combining = _class_body(lambda ch: unicodedata.category(ch) in ("Mn", "Mc"))
|
|
796
|
+
delta = _class_body(
|
|
797
|
+
lambda ch: ch.isalnum() and not ch.isalpha() and not ch.isdecimal())
|
|
798
|
+
upper_nonalpha = _class_body(lambda ch: ch.isupper() and not ch.isalpha())
|
|
799
|
+
return _Classes(
|
|
800
|
+
upper=upper, lower=lower, combining=combining, delta=delta,
|
|
801
|
+
upper_nonalpha=upper_nonalpha)
|
|
802
|
+
|
|
803
|
+
|
|
804
|
+
def uppercase_class() -> str:
|
|
805
|
+
"""Regex character-class body for the PREDICATE-signalling letters."""
|
|
806
|
+
return _classes().upper
|
|
807
|
+
|
|
808
|
+
|
|
809
|
+
def lowercase_class() -> str:
|
|
810
|
+
"""Regex character-class body for every term-valued letter (ordinary
|
|
811
|
+
lowercase, plus every letter of a script without case)."""
|
|
812
|
+
return _classes().lower
|
|
813
|
+
|
|
814
|
+
|
|
815
|
+
def combining_class() -> str:
|
|
816
|
+
"""Regex character-class body for combining marks (categories Mn/Mc)."""
|
|
817
|
+
return _classes().combining
|
|
818
|
+
|
|
819
|
+
|
|
820
|
+
# ---------------------------------------------------------------------------
|
|
821
|
+
# Small, fixed-size lookahead atoms — see the module docstring's "WHY THE
|
|
822
|
+
# GENERATED PATTERNS ARE SMALL" section for what these stand in for and why
|
|
823
|
+
# they are correct. Private: none of these is a character-class body (each
|
|
824
|
+
# opens with a lookahead assertion), so none of them can be spliced inside a
|
|
825
|
+
# caller's own ``[...]`` the way uppercase_class()/lowercase_class()/
|
|
826
|
+
# combining_class() can — they are used as bare regex atoms/alternatives
|
|
827
|
+
# only, exclusively by the pattern-builders below.
|
|
828
|
+
# ---------------------------------------------------------------------------
|
|
829
|
+
|
|
830
|
+
def _excluded_lookahead() -> str:
|
|
831
|
+
"""``(?!...)`` excluding :data:`_EXCLUDED_RANGES` — built from that same
|
|
832
|
+
tuple (never a second, hand-typed copy of the ranges), so it cannot
|
|
833
|
+
silently drift from what ``_classes()`` already excludes from
|
|
834
|
+
upper/lower/combining/delta.
|
|
835
|
+
|
|
836
|
+
Named for what it does rather than for Greek: since 0.23.2 the tuple also
|
|
837
|
+
carries the single-character operator glyphs (Ⓞ, Ⓖ, Ⓒ, …)."""
|
|
838
|
+
body = "".join(
|
|
839
|
+
_escape(lo) if lo == hi else f"{_escape(lo)}-{_escape(hi)}"
|
|
840
|
+
for lo, hi in _EXCLUDED_RANGES
|
|
841
|
+
)
|
|
842
|
+
return f"(?![{body}])"
|
|
843
|
+
|
|
844
|
+
|
|
845
|
+
def _delta_lookahead() -> str:
|
|
846
|
+
"""``(?!...)`` excluding :data:`_Classes.delta` — the ``\\w``-but-not-a-
|
|
847
|
+
letter sliver described in the module docstring's LOOKAHEAD section."""
|
|
848
|
+
return f"(?![{_classes().delta}])"
|
|
849
|
+
|
|
850
|
+
|
|
851
|
+
def _ceiling_lookahead() -> str:
|
|
852
|
+
"""``(?!...)`` excluding every codepoint above :data:`_MAX_CODEPOINT`.
|
|
853
|
+
|
|
854
|
+
``[^\\W\\d_]`` (Python's ``\\w`` minus digits/underscore) has no ceiling
|
|
855
|
+
of its own — ``\\w`` matches any codepoint with ``str.isalnum()`` true,
|
|
856
|
+
all the way to U+10FFFF, including scripts like CJK Unified Ideograph
|
|
857
|
+
Extension G (U+30000-U+3134A, category Lo) that :func:`_class_body`
|
|
858
|
+
never scans past ``_MAX_CODEPOINT`` to include. Without this lookahead
|
|
859
|
+
the LETTER atom would silently accept letters the fully-enumerated
|
|
860
|
+
``upper``/``lower`` classes it stands in for never could, breaking the
|
|
861
|
+
"differently-spelled pattern for the exact same set of strings" promise
|
|
862
|
+
this module's docstring makes for the lookahead rewrite. Built from
|
|
863
|
+
``_MAX_CODEPOINT`` itself (never a second, hand-typed boundary), so it
|
|
864
|
+
cannot drift out of sync with the ceiling every other class in this
|
|
865
|
+
module already respects."""
|
|
866
|
+
return f"(?![\\U{_MAX_CODEPOINT + 1:08x}-\\U0010ffff])"
|
|
867
|
+
|
|
868
|
+
|
|
869
|
+
@lru_cache(maxsize=1)
|
|
870
|
+
def _letter_atom() -> str:
|
|
871
|
+
"""LETTER: matches exactly one codepoint for which ``str.isalpha()`` OR
|
|
872
|
+
``str.isupper()`` holds (not ``isalpha()`` alone — see below), the
|
|
873
|
+
codepoint is not one of :data:`_EXCLUDED_RANGES`, and the codepoint is
|
|
874
|
+
at most :data:`_MAX_CODEPOINT` — i.e. exactly the union of
|
|
875
|
+
:func:`uppercase_class` and :func:`lowercase_class` (which is what "any
|
|
876
|
+
letter, either case" used to be built from directly, by literally
|
|
877
|
+
splicing both class bodies into a continuation position). Two
|
|
878
|
+
alternatives:
|
|
879
|
+
|
|
880
|
+
* ``[UPPER_NONALPHA]`` — the small explicit class of codepoints with
|
|
881
|
+
``isupper()`` true but ``isalpha()`` false (Roman numerals, circled/
|
|
882
|
+
squared Latin capitals; see :func:`_classes`'s ``upper_nonalpha``
|
|
883
|
+
docstring for why these need spelling out rather than a lookahead
|
|
884
|
+
tweak: some are not ``\\w`` members at all, so no negative lookahead
|
|
885
|
+
over ``\\w`` could ever admit them).
|
|
886
|
+
* ``(?!GREEK)(?!DELTA)(?!CEILING)[^\\W\\d_]`` — every ``isalpha()``
|
|
887
|
+
codepoint (upper or lower alike), which is everything ``uppercase_class()``
|
|
888
|
+
and ``lowercase_class()`` contain that is not already covered by the
|
|
889
|
+
first alternative.
|
|
890
|
+
|
|
891
|
+
Cached: it is pure string formatting over already-cached data, but it
|
|
892
|
+
gets referenced several times per terminal pattern across nine grammar
|
|
893
|
+
modes, so there is no reason to re-format it every time."""
|
|
894
|
+
upper_nonalpha = _classes().upper_nonalpha
|
|
895
|
+
isalpha_branch = (
|
|
896
|
+
f"{_excluded_lookahead()}{_delta_lookahead()}{_ceiling_lookahead()}"
|
|
897
|
+
f"[^\\W\\d_]"
|
|
898
|
+
)
|
|
899
|
+
return f"(?:[{upper_nonalpha}]|{isalpha_branch})"
|
|
900
|
+
|
|
901
|
+
|
|
902
|
+
@lru_cache(maxsize=1)
|
|
903
|
+
def _lowerish_atom() -> str:
|
|
904
|
+
"""The term-valued ("lowerish") letter: LETTER, with one more negative
|
|
905
|
+
lookahead excluding :func:`uppercase_class` stacked in front — i.e.
|
|
906
|
+
exactly what :func:`lowercase_class`'s body used to be spliced in for
|
|
907
|
+
directly, at every position a term-valued first character was needed.
|
|
908
|
+
Reuses UPPER's already-computed text as a lookahead rather than
|
|
909
|
+
re-deriving a second, separately-enumerated "letter minus upper"
|
|
910
|
+
class — see the module docstring's LOOKAHEAD section."""
|
|
911
|
+
return f"(?![{uppercase_class()}]){_letter_atom()}"
|
|
912
|
+
|
|
913
|
+
|
|
914
|
+
@lru_cache(maxsize=1)
|
|
915
|
+
def _continuation_atom(*, underscore: bool) -> str:
|
|
916
|
+
"""The shared "letter, digit, or combining mark" continuation atom,
|
|
917
|
+
optionally with ``_`` added. Every identifier terminal now passes
|
|
918
|
+
``underscore=True``; the flag survives because CONSTANT's ``c_`` form must
|
|
919
|
+
NOT (its own leading ``c_`` is the marker, and letting the tail carry more
|
|
920
|
+
underscores would widen the span it competes with NAME over — see the
|
|
921
|
+
module docstring's CONSTANT-priority section). Matches exactly one
|
|
922
|
+
character; callers append ``*``/``+`` themselves, the same way they would
|
|
923
|
+
to a character class — this is a non-capturing group standing in for one,
|
|
924
|
+
not a class body."""
|
|
925
|
+
digits = "0-9_" if underscore else "0-9"
|
|
926
|
+
return f"(?:{_letter_atom()}|[{digits}{combining_class()}])"
|
|
927
|
+
|
|
928
|
+
|
|
929
|
+
def predicate_pattern() -> str:
|
|
930
|
+
"""PREDICATE: an uppercase-signalling letter, then letters/digits/
|
|
931
|
+
underscores/combining marks — the SAME continuation class the term-valued
|
|
932
|
+
terminals get.
|
|
933
|
+
|
|
934
|
+
0.23.0 shipped this asymmetric, on the reasoning that keeping ``_`` out of
|
|
935
|
+
predicate position left an IRI-shaped name such as
|
|
936
|
+
``Http___www_w3_org_owl_Thing`` as illegal a predicate token as it had
|
|
937
|
+
always been, which ``fol/sanitize.py`` relies on. That reasoning does not
|
|
938
|
+
hold: ``sanitize.py`` carries its OWN deliberately ASCII-strict
|
|
939
|
+
``_PRED_RE`` (``[A-Z][a-zA-Z0-9]*``) and reaches its verdict without
|
|
940
|
+
consulting this module at all, so it renames that IRI either way.
|
|
941
|
+
|
|
942
|
+
What the asymmetry did break is ``chem/interop.py``. Its kit spelling of a
|
|
943
|
+
ChemLog predicate capitalises the FIRST character and nothing else, so 17
|
|
944
|
+
of the chemical signature's 40 predicates spell as ``Has_bond_to``,
|
|
945
|
+
``In_ring_of_size_6``, ``Net_charge_neutral`` and the like — names the kit's
|
|
946
|
+
own parser then refused, leaving the chemical vocabulary impossible to
|
|
947
|
+
write down in the kit's own surface syntax. A generating model handed the
|
|
948
|
+
signature and told to use it produced ``Has_bond_to(c, x)``, was refused,
|
|
949
|
+
and fell back to ``HasBondTo`` — which parses but is in no signature, so
|
|
950
|
+
every molecule came back as an uninterpreted-symbol error rather than a
|
|
951
|
+
verdict.
|
|
952
|
+
"""
|
|
953
|
+
cont = _continuation_atom(underscore=True)
|
|
954
|
+
return f"[{uppercase_class()}]{cont}*"
|
|
955
|
+
|
|
956
|
+
|
|
957
|
+
def name_pattern() -> str:
|
|
958
|
+
"""NAME: two alternatives, both term-valued.
|
|
959
|
+
|
|
960
|
+
The alpha-leading form starts with a term-valued letter, then any
|
|
961
|
+
continuation characters, then one more explicit letter (upper- or
|
|
962
|
+
lower-class — a predicate-signalling letter is legal HERE, mid-token;
|
|
963
|
+
only the first character carries the predicate/term distinction), then
|
|
964
|
+
more continuation — the same "at least two letters" shape the original
|
|
965
|
+
``[a-z][a-zA-Z0-9]*[a-zA-Z][a-zA-Z0-9]*`` had, just with underscores and
|
|
966
|
+
the wider alphabet spliced into every continuation run, so a single bare
|
|
967
|
+
letter still falls through to VARIABLE instead of NAME.
|
|
968
|
+
|
|
969
|
+
The digit-leading form is one-or-more ASCII digits, then a letter, then
|
|
970
|
+
the same continuation — see the module docstring's "WHY A DIGIT-LEADING
|
|
971
|
+
IDENTIFIER" section.
|
|
972
|
+
"""
|
|
973
|
+
cont = _continuation_atom(underscore=True)
|
|
974
|
+
letter = _letter_atom()
|
|
975
|
+
alpha_led = f"{_lowerish_atom()}{cont}*{letter}{cont}*"
|
|
976
|
+
digit_led = f"[0-9]+{letter}{cont}*"
|
|
977
|
+
return f"(?:{alpha_led})|(?:{digit_led})"
|
|
978
|
+
|
|
979
|
+
|
|
980
|
+
def variable_pattern() -> str:
|
|
981
|
+
"""VARIABLE: one term-valued letter, then only ASCII digits — unchanged
|
|
982
|
+
in shape from before this module, only the letter class is widened."""
|
|
983
|
+
return f"{_lowerish_atom()}[0-9]*"
|
|
984
|
+
|
|
985
|
+
|
|
986
|
+
def constant_pattern() -> str:
|
|
987
|
+
"""CONSTANT: the ``c_`` form (accepting Unicode letters/digits/combining
|
|
988
|
+
marks after the literal ``c_``, e.g. ``c_świątek``) or the plain lowercase
|
|
989
|
+
Greek run — untouched, since Greek is excluded from every generated class
|
|
990
|
+
(see the module docstring).
|
|
991
|
+
|
|
992
|
+
The ``c_`` form matches WHOLE WORDS only: it may not be followed by a
|
|
993
|
+
character that continues a NAME (a letter, digit, underscore or combining
|
|
994
|
+
mark). The lexer takes the first terminal that matches, not the longest, and
|
|
995
|
+
CONSTANT has the higher priority, so without the lookahead ``c_new_york``
|
|
996
|
+
was cut at ``c_new`` and the rest (``_york``) could not be read. With it,
|
|
997
|
+
CONSTANT declines the word and NAME reads all of it. The lookahead has to
|
|
998
|
+
name the whole continuation class and not just the underscore: a bare
|
|
999
|
+
``(?!_)`` lets the engine back off to ``c_ne`` and match that instead.
|
|
1000
|
+
"""
|
|
1001
|
+
cont = _continuation_atom(underscore=False)
|
|
1002
|
+
word_end = f"(?!{_continuation_atom(underscore=True)})"
|
|
1003
|
+
c_form = f"c_{cont}+{word_end}"
|
|
1004
|
+
greek_form = "[αβγδεζηθικνξοπρστυφχψω]+"
|
|
1005
|
+
return f"(?:{c_form})|(?:{greek_form})"
|
|
1006
|
+
|
|
1007
|
+
|
|
1008
|
+
def sort_pattern() -> str:
|
|
1009
|
+
"""SORT: a literal ``:`` then the same shape as PREDICATE — including the
|
|
1010
|
+
underscore, so a sorted signature can name a sort after the same vocabulary
|
|
1011
|
+
its predicates come from (``:In_ring``) instead of being the one identifier
|
|
1012
|
+
position that still cannot."""
|
|
1013
|
+
cont = _continuation_atom(underscore=True)
|
|
1014
|
+
return f":[{uppercase_class()}]{cont}*"
|
|
1015
|
+
|
|
1016
|
+
|
|
1017
|
+
def quoted_name_pattern() -> str:
|
|
1018
|
+
"""QUOTED_NAME: a constant written in single quotes.
|
|
1019
|
+
|
|
1020
|
+
A quote, then one or more of: any character that is not a quote, a backslash
|
|
1021
|
+
or one of the characters no spelling can carry (control characters, DEL,
|
|
1022
|
+
U+0085, U+2028, U+2029, surrogates: see :data:`_UNSPELLABLE_CLASS`), or the
|
|
1023
|
+
escape ``\\'`` (a quote) or ``\\\\`` (a backslash); then the closing quote. No
|
|
1024
|
+
other escape exists and the empty ``''`` is not a name.
|
|
1025
|
+
|
|
1026
|
+
The class is written with escapes (``\\x00``, ``\\u2028``) and not with the
|
|
1027
|
+
characters themselves: a literal line break inside a Lark ``/.../`` terminal
|
|
1028
|
+
is a grammar error, and a bare U+2028 in the pattern text would be invisible.
|
|
1029
|
+
No other terminal begins with a quote, so this one has no priority to win
|
|
1030
|
+
and none to lose.
|
|
1031
|
+
"""
|
|
1032
|
+
return f"'(?:[^'\\\\{_UNSPELLABLE_CLASS}]|\\\\['\\\\])+'"
|
|
1033
|
+
|
|
1034
|
+
|
|
1035
|
+
def terminal_block(*, include_sort: bool) -> str:
|
|
1036
|
+
"""The complete Lark terminal declarations for PREDICATE, CONSTANT, NAME,
|
|
1037
|
+
VARIABLE, QUOTED_NAME, and — when ``include_sort`` is true (the many-sorted
|
|
1038
|
+
modes) — SORT, in the priorities the grammar has always used
|
|
1039
|
+
(``CONSTANT.3`` over ``NAME.2`` over ``VARIABLE.1``; PREDICATE, QUOTED_NAME
|
|
1040
|
+
and SORT are unambiguous with everything else so carry no explicit
|
|
1041
|
+
priority). ``include_sort`` is a
|
|
1042
|
+
parameter rather than always-on because SORT never appears in a
|
|
1043
|
+
classical/modal/second-order grammar's rules, and declaring an unused
|
|
1044
|
+
terminal there is needless generated text for no behavioural gain."""
|
|
1045
|
+
lines = [
|
|
1046
|
+
f"PREDICATE: /{predicate_pattern()}/",
|
|
1047
|
+
"",
|
|
1048
|
+
f"CONSTANT.3: /{constant_pattern()}/",
|
|
1049
|
+
"",
|
|
1050
|
+
f"NAME.2: /{name_pattern()}/",
|
|
1051
|
+
"",
|
|
1052
|
+
f"VARIABLE.1: /{variable_pattern()}/",
|
|
1053
|
+
"",
|
|
1054
|
+
f"QUOTED_NAME: /{quoted_name_pattern()}/",
|
|
1055
|
+
]
|
|
1056
|
+
if include_sort:
|
|
1057
|
+
lines += ["", f"SORT: /{sort_pattern()}/"]
|
|
1058
|
+
return "\n".join(lines) + "\n"
|
|
1059
|
+
|
|
1060
|
+
|
|
1061
|
+
#: Terminal name -> short English description of its shape, for
|
|
1062
|
+
#: ``fol/naming.py``'s NamingError to show in place of the generated regex
|
|
1063
|
+
#: text (a few kilobytes of ``\uXXXX-\uYYYY`` ranges is not a message a
|
|
1064
|
+
#: human — or a model reading the error to retry — can act on).
|
|
1065
|
+
HUMAN_READABLE_PATTERNS = {
|
|
1066
|
+
"PREDICATE": (
|
|
1067
|
+
"an uppercase letter (in any script that has letter case) followed "
|
|
1068
|
+
"by letters, digits, or combining marks"
|
|
1069
|
+
),
|
|
1070
|
+
"NAME": (
|
|
1071
|
+
"a lowercase or caseless letter, then letters/digits/underscores/"
|
|
1072
|
+
"combining marks with at least one more letter among them; or one "
|
|
1073
|
+
"or more digits followed by a letter and more of the same"
|
|
1074
|
+
),
|
|
1075
|
+
"CONSTANT": (
|
|
1076
|
+
"'c_' followed by letters, digits, or combining marks, or one or "
|
|
1077
|
+
"more lowercase Greek letters (α-ω)"
|
|
1078
|
+
),
|
|
1079
|
+
"QUOTED_NAME": (
|
|
1080
|
+
"a single quote, then the name, then a single quote; inside, a "
|
|
1081
|
+
"quote is written \\' and a backslash \\\\"
|
|
1082
|
+
),
|
|
1083
|
+
"VARIABLE": (
|
|
1084
|
+
"a single lowercase or caseless letter, optionally followed by "
|
|
1085
|
+
"digits"
|
|
1086
|
+
),
|
|
1087
|
+
"SORT": (
|
|
1088
|
+
"':' followed by an uppercase letter and then letters, digits, or "
|
|
1089
|
+
"combining marks"
|
|
1090
|
+
),
|
|
1091
|
+
}
|