unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,391 @@
|
|
|
1
|
+
"""A string parser for ALC(Q) concepts, dual to :meth:`Concept.to_unicode`.
|
|
2
|
+
|
|
3
|
+
``dl.concepts`` builds concepts only by Python construction (``dl.And(A,
|
|
4
|
+
dl.Not(B))``); this module parses the glyph syntax that
|
|
5
|
+
:meth:`~unicode_logic_kit.dl.concepts.Concept.to_unicode` emits back into a
|
|
6
|
+
:class:`~unicode_logic_kit.dl.concepts.Concept`, so a rendered concept — or one
|
|
7
|
+
typed by hand in the same notation — round-trips: ``parse_concept(c.to_unicode())
|
|
8
|
+
== c`` for every constructor, including the qualified number restrictions
|
|
9
|
+
``AtLeast``/``AtMost`` (≥n r.C / ≤n r.C), with the documented exceptions below.
|
|
10
|
+
|
|
11
|
+
The reader folds a flat chain of one connective to the LEFT (``A ⊓ B ⊓ C`` is
|
|
12
|
+
``(A ⊓ B) ⊓ C``), so the printer writes a nested operand of the SAME connective
|
|
13
|
+
without parentheses on the left only: ``And(A, And(B, C))`` is ``A ⊓ (B ⊓ C)``,
|
|
14
|
+
never ``A ⊓ B ⊓ C``, which reads back as the other tree. The round trip is
|
|
15
|
+
therefore exact for every shape, not only for left-nested chains.
|
|
16
|
+
|
|
17
|
+
The first exceptions are the two constructors that name an INDIVIDUAL:
|
|
18
|
+
:class:`~unicode_logic_kit.dl.concepts.Nominal` renders as ``{a}`` and
|
|
19
|
+
:class:`~unicode_logic_kit.dl.concepts.HasValue` as ``∃r.{a}``, and neither reads
|
|
20
|
+
back — they are refused BY NAME instead (see ``_primary``). The glyph syntax
|
|
21
|
+
has no individual-name layer, so ``{a}`` cannot be told apart from a concept
|
|
22
|
+
NAME, and ``∃r.{a}`` cannot be told apart from ``∃r.`` applied to a nominal:
|
|
23
|
+
``HasValue("r", "a")`` and ``Exists("r", Nominal("a"))`` have the SAME
|
|
24
|
+
rendering, honestly, because they have the same models — but picking one of
|
|
25
|
+
them when reading the text back would be a silent normalisation of the other.
|
|
26
|
+
This is the same render-only asymmetry :func:`~unicode_logic_kit.dl.to_manchester`
|
|
27
|
+
already has for a nominal, and the two syntaxes that DO distinguish the pair,
|
|
28
|
+
Manchester (``r value a`` versus ``r some {a}``) and Functional
|
|
29
|
+
(``ObjectHasValue(r a)`` versus ``ObjectSomeValuesFrom(r ObjectOneOf(a))``),
|
|
30
|
+
round-trip it exactly.
|
|
31
|
+
|
|
32
|
+
The DATA restrictions (:class:`~unicode_logic_kit.dl.concepts.DataExists` and its
|
|
33
|
+
four siblings) are render-only in the same way, for the same reason: the glyph
|
|
34
|
+
syntax has no data-range layer, and ``∃d.xsd:integer`` is the text of BOTH
|
|
35
|
+
``DataExists("d", Datatype("xsd:integer"))`` and ``Exists("d",
|
|
36
|
+
Atomic("xsd:integer"))`` -- two concepts about different sorts. Reading it as
|
|
37
|
+
the second would be the silent misreading ``dl.parse_manchester`` had for
|
|
38
|
+
``d some xsd:integer`` before the data layer. So a NAME that is a BUILT-IN
|
|
39
|
+
datatype (``xsd:integer``, ``rdfs:Literal``, …) is refused by name below; OWL 2
|
|
40
|
+
forbids a class with a datatype's name anyway. A USER-defined datatype
|
|
41
|
+
(``∃d.Digit``) cannot be told from a class by its spelling and reads as the
|
|
42
|
+
object restriction -- the limit the Manchester reader avoids by being told the
|
|
43
|
+
datatype names (``datatypes=``). Read data restrictions with
|
|
44
|
+
``dl.parse_manchester`` or ``dl.parse_owl_functional_class_expression``.
|
|
45
|
+
|
|
46
|
+
An :class:`~unicode_logic_kit.dl.concepts.InverseRole` is the last one: it prints
|
|
47
|
+
as ``r⁻`` (``∃r⁻.C``), and the reader refuses a role name that ends in ``⁻`` BY
|
|
48
|
+
NAME (see ``_role_name``) instead of reading a plain role called ``r⁻``, which
|
|
49
|
+
would turn an inverse role into an unrelated one without a word.
|
|
50
|
+
|
|
51
|
+
Names. The grammar below says which names the reader can read. A name outside
|
|
52
|
+
it (one that holds whitespace, an ``.`` or a parenthesis, one of the reserved
|
|
53
|
+
glyphs, or nothing at all) is printed as it is, for display, and the reader then
|
|
54
|
+
refuses the text: this syntax has no escape (the Manchester and Functional-Style
|
|
55
|
+
writers bracket such a name as a full IRI). The one thing the printer does not do
|
|
56
|
+
is print a name so that the text reads back as ANOTHER concept — ``Atomic("A ⊓ B")``
|
|
57
|
+
would print ``A ⊓ B``, the intersection of two classes — and
|
|
58
|
+
``Concept.to_unicode`` raises :class:`ValueError` for it.
|
|
59
|
+
|
|
60
|
+
Grammar (loosest-binding first, matching ``concepts.py``'s ``_PREC`` table
|
|
61
|
+
exactly — ⊔ at precedence 1, ⊓ at 2, ¬/∃/∀/≥/≤ at 3, atoms/⊤/⊥ at 4)::
|
|
62
|
+
|
|
63
|
+
concept := or
|
|
64
|
+
or := and ("⊔" and)*
|
|
65
|
+
and := unary ("⊓" unary)*
|
|
66
|
+
unary := "¬" unary
|
|
67
|
+
| ("∃" | "∀") NAME "." unary
|
|
68
|
+
| ("≥" | "≤") NUMBER NAME "." unary
|
|
69
|
+
| primary
|
|
70
|
+
primary := "⊤" | "⊥" | NAME | "(" concept ")"
|
|
71
|
+
NAME := a maximal run of characters that are none of: whitespace,
|
|
72
|
+
the glyphs ⊤ ⊥ ¬ ⊓ ⊔ ∃ ∀ ≥ ≤ ( ) . ⊑ { }
|
|
73
|
+
NUMBER := a NAME token consisting only of ASCII digits
|
|
74
|
+
|
|
75
|
+
A concept/role NAME may be any length and contain any characters outside
|
|
76
|
+
that reserved set (digits, underscores, non-ASCII letters, …) — e.g.
|
|
77
|
+
``hasChild``, ``Doctor42``, ``θ``. ``{`` and ``}`` are in that reserved set
|
|
78
|
+
since 0.30.0 and are refused BY NAME (see ``_primary``): a nominal-shaped
|
|
79
|
+
text like ``{a}`` was a legal NAME before, so ``parse_concept("{a}")``
|
|
80
|
+
returned ``Atomic("{a}")`` -- a concept with a bogus class name -- silently,
|
|
81
|
+
for text ``dl.concepts`` itself prints. A concept name containing a literal
|
|
82
|
+
brace therefore stops parsing; there is no such name in the kit, its tests or
|
|
83
|
+
the OEO ontology this was measured against.
|
|
84
|
+
Restricting ``unary``'s operand (rather
|
|
85
|
+
than the full ``concept``) is what makes ``∃r.∀s.C`` and ``¬¬C`` parse
|
|
86
|
+
without parentheses while ``∃r.(C ⊓ D)`` requires them, mirroring
|
|
87
|
+
``concepts.py``'s ``_paren`` exactly — so the grammar is precedence-faithful
|
|
88
|
+
by construction, not just by testing. ``≥``/``≤`` need a NUMBER token (the
|
|
89
|
+
bound ``n``) between the glyph and the role name, rendered by
|
|
90
|
+
``Concept.to_unicode`` with a mandatory space before the role name (``"≥2
|
|
91
|
+
r.C"``, never ``"≥2r.C"``) so the tokenizer — which has no notion of a
|
|
92
|
+
digit/letter boundary — can always split the two apart; see ``_unary`` below.
|
|
93
|
+
|
|
94
|
+
This module intentionally implements a small hand-rolled recursive-descent
|
|
95
|
+
parser rather than reusing the Lark-based registry machinery in
|
|
96
|
+
``fol/msflparser.py``: that machinery is purpose-built for assembling several
|
|
97
|
+
large, mutually-exclusive FOL dialects (FOL/MSFOL/MSFL/modal/…) that share a
|
|
98
|
+
term/atom/lambda layer, none of which applies here — ALC concepts are a tiny,
|
|
99
|
+
fixed, single-purpose grammar with no term layer, no dialects, and no
|
|
100
|
+
sharing to gain from the registry. A ~150-line hand-rolled parser is both
|
|
101
|
+
simpler and easier to audit for this shape of grammar than standing up a
|
|
102
|
+
Lark grammar file plus transformer for it.
|
|
103
|
+
|
|
104
|
+
Glyph-only: there is no ASCII fallback syntax (no ``E`` for ``∃``, ``A`` for
|
|
105
|
+
``∀``, ``&`` for ``⊓``, …). Unlike the operator glyphs, ``A`` and ``E`` are
|
|
106
|
+
exactly the kind of short identifiers real concept/role names use (as they
|
|
107
|
+
do throughout ``tests/test_dl_alc.py``), so ASCII keyword fallbacks for the
|
|
108
|
+
quantifier-like operators would silently swallow a large fraction of
|
|
109
|
+
plausible concept names — not "trivially cheap" — so they are not provided.
|
|
110
|
+
|
|
111
|
+
GCIs (general concept inclusions, ``C ⊑ D``): ``concepts.py`` has no
|
|
112
|
+
renderer for them (:class:`~unicode_logic_kit.dl.tableau.TBox` carries
|
|
113
|
+
``(Concept, Concept)`` pairs with no ``to_unicode``), so there is no
|
|
114
|
+
round-trip counterpart to test :func:`parse_gci` against — it is provided as
|
|
115
|
+
a convenience for hand-written GCI text using ⊑, the glyph the ``TBox`` /
|
|
116
|
+
``subsumes`` docstrings already use for this notion, and is verified directly
|
|
117
|
+
against hand-picked expected ``(Concept, Concept)`` pairs instead.
|
|
118
|
+
"""
|
|
119
|
+
|
|
120
|
+
from typing import List, Tuple
|
|
121
|
+
|
|
122
|
+
from .datatypes import is_builtin_datatype
|
|
123
|
+
from .concepts import (
|
|
124
|
+
Concept, Top, Bottom, Atomic, Not, And, Or, Exists, ForAll, AtLeast, AtMost,
|
|
125
|
+
)
|
|
126
|
+
from .tableau import RoleExpressionError, _reject_concept_role
|
|
127
|
+
|
|
128
|
+
__all__ = ["parse_concept", "parse_gci", "ConceptSyntaxError"]
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
class ConceptSyntaxError(ValueError):
|
|
132
|
+
"""Raised by :func:`parse_concept` / :func:`parse_gci` on malformed input."""
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
# --------------------------------------------------------------------------- #
|
|
136
|
+
# Tokenizer.
|
|
137
|
+
# --------------------------------------------------------------------------- #
|
|
138
|
+
|
|
139
|
+
_GLYPH_TOKENS = {
|
|
140
|
+
"⊤": "TOP", "⊥": "BOT", "¬": "NOT", "⊓": "AND", "⊔": "OR",
|
|
141
|
+
"∃": "EXISTS", "∀": "FORALL", "≥": "ATLEAST", "≤": "ATMOST",
|
|
142
|
+
"(": "LPAREN", ")": "RPAREN", ".": "DOT", "⊑": "SUBSUME",
|
|
143
|
+
# RESERVED since 0.30.0, so `{a}` is refused BY NAME instead of read as a
|
|
144
|
+
# concept NAME spelled "{a}" -- see _primary's LBRACE branch.
|
|
145
|
+
"{": "LBRACE", "}": "RBRACE",
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
# A Token is (type: str, value: str, pos: int).
|
|
149
|
+
_Token = Tuple[str, str, int]
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _tokenize(text: str) -> List[_Token]:
|
|
153
|
+
"""Split ``text`` into glyph tokens and maximal-run NAME tokens, plus a
|
|
154
|
+
trailing EOF sentinel (whose ``pos`` is ``len(text)``, for error messages).
|
|
155
|
+
"""
|
|
156
|
+
tokens: List[_Token] = []
|
|
157
|
+
i, n = 0, len(text)
|
|
158
|
+
while i < n:
|
|
159
|
+
ch = text[i]
|
|
160
|
+
if ch.isspace():
|
|
161
|
+
i += 1
|
|
162
|
+
continue
|
|
163
|
+
if ch in _GLYPH_TOKENS:
|
|
164
|
+
tokens.append((_GLYPH_TOKENS[ch], ch, i))
|
|
165
|
+
i += 1
|
|
166
|
+
continue
|
|
167
|
+
start = i
|
|
168
|
+
while i < n and not text[i].isspace() and text[i] not in _GLYPH_TOKENS:
|
|
169
|
+
i += 1
|
|
170
|
+
tokens.append(("NAME", text[start:i], start))
|
|
171
|
+
tokens.append(("EOF", "", n))
|
|
172
|
+
return tokens
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
# --------------------------------------------------------------------------- #
|
|
176
|
+
# Recursive-descent parser.
|
|
177
|
+
# --------------------------------------------------------------------------- #
|
|
178
|
+
|
|
179
|
+
class _Parser:
|
|
180
|
+
"""A single parse of one token stream; not re-used across calls."""
|
|
181
|
+
|
|
182
|
+
def __init__(self, tokens: List[_Token], text: str):
|
|
183
|
+
self._tokens = tokens
|
|
184
|
+
self._text = text
|
|
185
|
+
self._i = 0
|
|
186
|
+
|
|
187
|
+
def _peek(self) -> _Token:
|
|
188
|
+
return self._tokens[self._i]
|
|
189
|
+
|
|
190
|
+
def _advance(self) -> _Token:
|
|
191
|
+
tok = self._tokens[self._i]
|
|
192
|
+
self._i += 1
|
|
193
|
+
return tok
|
|
194
|
+
|
|
195
|
+
def _error(self, message: str) -> ConceptSyntaxError:
|
|
196
|
+
return ConceptSyntaxError(f"{message} in {self._text!r}")
|
|
197
|
+
|
|
198
|
+
def _expect(self, ttype: str, what: str) -> _Token:
|
|
199
|
+
tok = self._peek()
|
|
200
|
+
if tok[0] != ttype:
|
|
201
|
+
found = "end of input" if tok[0] == "EOF" else f"{tok[1]!r}"
|
|
202
|
+
raise self._error(
|
|
203
|
+
f"parse_concept: expected {what} but found {found} "
|
|
204
|
+
f"at position {tok[2]}")
|
|
205
|
+
return self._advance()
|
|
206
|
+
|
|
207
|
+
def _expect_eof(self) -> None:
|
|
208
|
+
tok = self._peek()
|
|
209
|
+
if tok[0] != "EOF":
|
|
210
|
+
raise self._error(
|
|
211
|
+
f"parse_concept: unexpected trailing input {tok[1]!r} "
|
|
212
|
+
f"at position {tok[2]}")
|
|
213
|
+
|
|
214
|
+
# -- grammar levels, loosest first (mirrors concepts.py's _PREC) -------- #
|
|
215
|
+
|
|
216
|
+
def _or(self) -> Concept:
|
|
217
|
+
left = self._and()
|
|
218
|
+
while self._peek()[0] == "OR":
|
|
219
|
+
self._advance()
|
|
220
|
+
left = Or(left, self._and())
|
|
221
|
+
return left
|
|
222
|
+
|
|
223
|
+
def _and(self) -> Concept:
|
|
224
|
+
left = self._unary()
|
|
225
|
+
while self._peek()[0] == "AND":
|
|
226
|
+
self._advance()
|
|
227
|
+
left = And(left, self._unary())
|
|
228
|
+
return left
|
|
229
|
+
|
|
230
|
+
def _checked(self, concept: Concept, pos: int) -> Concept:
|
|
231
|
+
"""``concept``, unless its role is an OWL 2 built-in property name (or
|
|
232
|
+
``=``/``≠``) — refused BY NAME, the way ``dl.parse_owl_functional`` and
|
|
233
|
+
the Manchester parser refuse it. ``∃owl:topObjectProperty.A`` used to
|
|
234
|
+
read as an ordinary role of that name, whose verdict (satisfiable) is
|
|
235
|
+
not the universal property's (every element is related to itself)."""
|
|
236
|
+
try:
|
|
237
|
+
_reject_concept_role(concept, where="parse_concept")
|
|
238
|
+
except RoleExpressionError as exc:
|
|
239
|
+
raise self._error(f"{exc} (at position {pos})") from exc
|
|
240
|
+
return concept
|
|
241
|
+
|
|
242
|
+
def _unary(self) -> Concept:
|
|
243
|
+
ttype, _, pos = self._peek()
|
|
244
|
+
if ttype == "NOT":
|
|
245
|
+
self._advance()
|
|
246
|
+
return Not(self._unary())
|
|
247
|
+
if ttype in ("EXISTS", "FORALL"):
|
|
248
|
+
self._advance()
|
|
249
|
+
role = self._role_name()
|
|
250
|
+
self._expect("DOT", "'.' after the role name")
|
|
251
|
+
body = self._unary()
|
|
252
|
+
return self._checked(
|
|
253
|
+
Exists(role, body) if ttype == "EXISTS" else ForAll(role, body), pos)
|
|
254
|
+
if ttype in ("ATLEAST", "ATMOST"):
|
|
255
|
+
self._advance()
|
|
256
|
+
n = self._expect_number()
|
|
257
|
+
role = self._role_name()
|
|
258
|
+
self._expect("DOT", "'.' after the role name")
|
|
259
|
+
body = self._unary()
|
|
260
|
+
return self._checked(
|
|
261
|
+
AtLeast(n, role, body) if ttype == "ATLEAST" else AtMost(n, role, body),
|
|
262
|
+
pos)
|
|
263
|
+
return self._primary()
|
|
264
|
+
|
|
265
|
+
def _role_name(self) -> str:
|
|
266
|
+
"""Consume the NAME token of a restriction's role.
|
|
267
|
+
|
|
268
|
+
A name that ends in the inverse glyph ``⁻`` is refused BY NAME: that is
|
|
269
|
+
how ``Concept.to_unicode`` writes ``Exists(InverseRole("r"), C)``
|
|
270
|
+
(``∃r⁻.C``), a role EXPRESSION this syntax has no layer for, so reading
|
|
271
|
+
the text as a role NAMED ``r⁻`` would silently turn an inverse role into
|
|
272
|
+
an unrelated plain one.
|
|
273
|
+
"""
|
|
274
|
+
tok = self._expect("NAME", "a role name")
|
|
275
|
+
if tok[1].endswith("⁻"):
|
|
276
|
+
raise self._error(
|
|
277
|
+
f"parse_concept: {tok[1]!r} (position {tok[2]}) is the glyph "
|
|
278
|
+
f"spelling of an INVERSE role, which the glyph syntax cannot "
|
|
279
|
+
f"read — reading it as a role named {tok[1]!r} would silently "
|
|
280
|
+
f"change what it says. Build dl.InverseRole({tok[1][:-1]!r}) "
|
|
281
|
+
f"directly (the in-house tableau refuses it by name; the "
|
|
282
|
+
f"external reasoner, dl.owl_reasoner, decides it)")
|
|
283
|
+
return tok[1]
|
|
284
|
+
|
|
285
|
+
def _expect_number(self) -> int:
|
|
286
|
+
"""Consume a NAME token of only ASCII digits (the ``n`` in ``≥n``/``≤n``)."""
|
|
287
|
+
tok = self._peek()
|
|
288
|
+
if tok[0] == "NAME" and tok[1].isdigit():
|
|
289
|
+
self._advance()
|
|
290
|
+
return int(tok[1])
|
|
291
|
+
found = "end of input" if tok[0] == "EOF" else f"{tok[1]!r}"
|
|
292
|
+
raise self._error(
|
|
293
|
+
f"parse_concept: expected a non-negative integer but found {found} "
|
|
294
|
+
f"at position {tok[2]}")
|
|
295
|
+
|
|
296
|
+
def _primary(self) -> Concept:
|
|
297
|
+
ttype, value, pos = self._peek()
|
|
298
|
+
if ttype == "TOP":
|
|
299
|
+
self._advance()
|
|
300
|
+
return Top()
|
|
301
|
+
if ttype == "BOT":
|
|
302
|
+
self._advance()
|
|
303
|
+
return Bottom()
|
|
304
|
+
if ttype in ("LBRACE", "RBRACE"):
|
|
305
|
+
# A nominal-shaped text. Until 0.30.0 braces were not reserved, so
|
|
306
|
+
# `{a}` was a legal NAME and parse_concept("{a}") returned
|
|
307
|
+
# Atomic("{a}") -- a concept with a bogus class name, silently, for
|
|
308
|
+
# text this module's own siblings print. dl.parse_manchester
|
|
309
|
+
# already refused the same text by name; now so does this.
|
|
310
|
+
#
|
|
311
|
+
# Teaching it to BUILD a nominal was the alternative and is wrong:
|
|
312
|
+
# `∃r.{a}` is AMBIGUOUS between dl.HasValue(r, "a") and
|
|
313
|
+
# dl.Exists(r, dl.Nominal("a")) -- the glyph syntax has no
|
|
314
|
+
# individual-name layer to tell them apart -- and always picking
|
|
315
|
+
# one would be a silent normalisation of the other.
|
|
316
|
+
raise self._error(
|
|
317
|
+
f"parse_concept: nominal and value concepts "
|
|
318
|
+
f"('{{a}}', '∃r.{{a}}') are not supported — the glyph syntax "
|
|
319
|
+
f"has no individual-name layer, so '{{a}}' cannot be told "
|
|
320
|
+
f"apart from a concept NAME, and '∃r.{{a}}' cannot be told "
|
|
321
|
+
f"apart from a value restriction (found {value!r} at position "
|
|
322
|
+
f"{pos}). Build dl.Nominal('a') or dl.HasValue(role, 'a') "
|
|
323
|
+
f"directly, or read 'r value a' with dl.parse_manchester / "
|
|
324
|
+
f"'ObjectHasValue(r a)' with "
|
|
325
|
+
f"dl.parse_owl_functional_class_expression")
|
|
326
|
+
if ttype == "NAME":
|
|
327
|
+
if is_builtin_datatype(value):
|
|
328
|
+
raise self._error(
|
|
329
|
+
f"parse_concept: {value!r} (position {pos}) is a built-in "
|
|
330
|
+
f"DATATYPE, not a class, and the glyph syntax has no "
|
|
331
|
+
f"data-range layer -- reading '∃d.{value}' as an object "
|
|
332
|
+
f"restriction would silently change what it says. Read a "
|
|
333
|
+
f"data restriction with dl.parse_manchester "
|
|
334
|
+
f"('d some {value}') or "
|
|
335
|
+
f"dl.parse_owl_functional_class_expression "
|
|
336
|
+
f"('DataSomeValuesFrom(d {value})')")
|
|
337
|
+
self._advance()
|
|
338
|
+
return Atomic(value)
|
|
339
|
+
if ttype == "LPAREN":
|
|
340
|
+
self._advance()
|
|
341
|
+
inner = self._or()
|
|
342
|
+
self._expect("RPAREN", "')'")
|
|
343
|
+
return inner
|
|
344
|
+
found = "end of input" if ttype == "EOF" else f"{value!r}"
|
|
345
|
+
raise self._error(
|
|
346
|
+
f"parse_concept: unexpected {found} at position {pos}; expected "
|
|
347
|
+
"'⊤', '⊥', a concept name, '¬', '∃', '∀', '≥', '≤', or '('")
|
|
348
|
+
|
|
349
|
+
# -- entry points --------------------------------------------------- #
|
|
350
|
+
|
|
351
|
+
def parse_concept(self) -> Concept:
|
|
352
|
+
c = self._or()
|
|
353
|
+
self._expect_eof()
|
|
354
|
+
return c
|
|
355
|
+
|
|
356
|
+
def parse_gci(self) -> Tuple[Concept, Concept]:
|
|
357
|
+
sub = self._or()
|
|
358
|
+
self._expect("SUBSUME", "'⊑'")
|
|
359
|
+
sup = self._or()
|
|
360
|
+
self._expect_eof()
|
|
361
|
+
return sub, sup
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
def parse_concept(text: str) -> Concept:
|
|
365
|
+
"""Parse ``text`` (the ⊤ ⊥ ¬ ⊓ ⊔ ∃ ∀ ≥ ≤ glyph syntax) into a :class:`Concept`.
|
|
366
|
+
|
|
367
|
+
Round-trips against :meth:`Concept.to_unicode`: ``parse_concept(c.to_unicode())
|
|
368
|
+
== c`` for every concept ``c`` EXCEPT the ones the glyph syntax cannot read:
|
|
369
|
+
the two that name an individual
|
|
370
|
+
(:class:`~unicode_logic_kit.dl.concepts.Nominal` and
|
|
371
|
+
:class:`~unicode_logic_kit.dl.concepts.HasValue`, which render as ``{a}`` and
|
|
372
|
+
``∃r.{a}`` and are refused by name on the way back in — see the module
|
|
373
|
+
docstring and :class:`~unicode_logic_kit.dl.concepts.HasValue`; the same
|
|
374
|
+
render-only asymmetry ``dl.to_manchester`` has for a nominal), the data
|
|
375
|
+
restrictions, and an inverse role (``∃r⁻.C``), all refused by name. The
|
|
376
|
+
shape of a chain is exact: ``And(A, And(B, C))`` is written ``A ⊓ (B ⊓ C)``.
|
|
377
|
+
Raises :class:`ConceptSyntaxError` on
|
|
378
|
+
malformed input (unbalanced parentheses, a missing '.' after a role name,
|
|
379
|
+
a stray operator, trailing garbage, …).
|
|
380
|
+
"""
|
|
381
|
+
return _Parser(_tokenize(text), text).parse_concept()
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
def parse_gci(text: str) -> Tuple[Concept, Concept]:
|
|
385
|
+
"""Parse a general concept inclusion ``"C ⊑ D"`` into ``(C, D)``.
|
|
386
|
+
|
|
387
|
+
See the module docstring for why this has no round-trip counterpart to
|
|
388
|
+
test against (``concepts.py`` never renders a GCI). Raises
|
|
389
|
+
:class:`ConceptSyntaxError` on malformed input.
|
|
390
|
+
"""
|
|
391
|
+
return _Parser(_tokenize(text), text).parse_gci()
|