unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,1637 @@
|
|
|
1
|
+
"""A parser/renderer for the **OWL 2 Manchester Syntax**, restricted to ALCHQ.
|
|
2
|
+
|
|
3
|
+
`OWL 2 Manchester Syntax <https://www.w3.org/TR/owl2-manchester-syntax/>`_ is
|
|
4
|
+
the W3C-standardised, keyword-based (as opposed to :mod:`unicode_logic_kit.dl.parser`'s
|
|
5
|
+
glyph-based) concrete syntax for OWL 2 class expressions — the notation Protégé
|
|
6
|
+
and most OWL tooling show by default (``Person and hasChild some Doctor``
|
|
7
|
+
rather than ``Person ⊓ ∃hasChild.Doctor``). This module is the *OWL bridge,
|
|
8
|
+
part 1*: it parses (and renders) exactly the ALCHQ-expressible fragment of that
|
|
9
|
+
syntax into/from the kit's existing :class:`~unicode_logic_kit.dl.concepts.Concept`
|
|
10
|
+
AST, so any Manchester-syntax ALCHQ expression becomes reachable by every kit
|
|
11
|
+
reasoner and export (:mod:`unicode_logic_kit.dl.tableau`,
|
|
12
|
+
:mod:`unicode_logic_kit.dl.translate`, …) with no separate code path.
|
|
13
|
+
|
|
14
|
+
Supported fragment (ALCHQ, matching :mod:`unicode_logic_kit.dl.concepts` plus the
|
|
15
|
+
role-hierarchy/transitivity RBox axioms below):
|
|
16
|
+
|
|
17
|
+
- class names (``Person``, ``hasChild`` as a role name);
|
|
18
|
+
- ``C and D``, ``C or D``, ``not C``;
|
|
19
|
+
- ``r some C`` (∃r.C), ``r only C`` (∀r.C);
|
|
20
|
+
- ``r value a`` (∃r.{a}) — a value restriction, mapping onto
|
|
21
|
+
:class:`~unicode_logic_kit.dl.concepts.HasValue`, a concept kind of its
|
|
22
|
+
own and NOT ``r some {a}``, which stays refused (see that class's
|
|
23
|
+
docstring for why the two must not collapse into one AST shape);
|
|
24
|
+
- ``r min n C`` / ``r max n C`` / ``r exactly n C`` (≥n r.C / ≤n r.C /
|
|
25
|
+
≥n r.C ⊓ ≤n r.C) — the qualifying class ``C`` is OPTIONAL per the W3C
|
|
26
|
+
grammar and defaults to ``owl:Thing`` when omitted (``r min n`` alone);
|
|
27
|
+
see "Qualified number restrictions" below;
|
|
28
|
+
- parentheses;
|
|
29
|
+
- ``owl:Thing`` / ``owl:Nothing`` for ⊤ / ⊥ (the W3C grammar treats these
|
|
30
|
+
as ordinary ``owl:``-prefixed class names, not keywords — see "Top and
|
|
31
|
+
Bottom" below — and this module maps exactly those two spellings onto
|
|
32
|
+
:class:`~unicode_logic_kit.dl.concepts.Top` / :class:`~unicode_logic_kit.dl.concepts.Bottom`).
|
|
33
|
+
|
|
34
|
+
Rejected (real Manchester/OWL 2 syntax, but outside ALCHQ — see "Rejected
|
|
35
|
+
constructs" below): ``Self`` restrictions, inverse
|
|
36
|
+
roles (``inverse r``), nominal ("one-of") concepts (``{a, b}``), and the facets
|
|
37
|
+
of a datatype restriction that have no first-order image
|
|
38
|
+
(``xsd:pattern``, the length facets, an ordering facet on a non-numeric base).
|
|
39
|
+
Each is rejected with a
|
|
40
|
+
:class:`ManchesterSyntaxError` naming the specific construct, per the kit's
|
|
41
|
+
honesty convention (an unsupported construct is a loud, precise error —
|
|
42
|
+
never a silent mistranslation or a weaker-than-requested result).
|
|
43
|
+
|
|
44
|
+
Data restrictions
|
|
45
|
+
------------------
|
|
46
|
+
A restriction whose filler is a DATA RANGE is read as a data restriction
|
|
47
|
+
(:class:`~unicode_logic_kit.dl.concepts.DataExists` and friends), not as an
|
|
48
|
+
object restriction with a class that happens to be called ``xsd:integer``:
|
|
49
|
+
|
|
50
|
+
* ``d some xsd:integer``, ``d only DR`` -> ``DataExists`` / ``DataForAll``;
|
|
51
|
+
* ``d min 2 DR`` / ``d max 2 DR`` / ``d exactly 2 DR`` -> ``DataAtLeast`` /
|
|
52
|
+
``DataAtMost`` (``exactly`` as their conjunction, as for objects);
|
|
53
|
+
* ``d value "1"^^xsd:integer`` (or a bare numeral, ``d value 1``) ->
|
|
54
|
+
``DataHasValue``. The BARE numerals read are the W3C grammar's three
|
|
55
|
+
unquoted literal forms: an integer (``1``, ``-3``, an ``xsd:integer``), a
|
|
56
|
+
decimal (``1.5``, an ``xsd:decimal``) and a floating-point literal, which is
|
|
57
|
+
an ``xsd:float`` and ends in ``f`` or ``F`` (``1.5f``, ``2F``, ``-.5f``,
|
|
58
|
+
``1.0e+2f``). The data layer has no first-order image of ``xsd:float`` (its
|
|
59
|
+
value space has ``NaN`` and a signed zero), so ``d value 1.5f`` is the same
|
|
60
|
+
``DataHasValue`` as ``d value "1.5"^^xsd:float`` and is refused BY NAME at
|
|
61
|
+
translation, exactly as that quoted spelling and the Functional-Style
|
|
62
|
+
``"1.5"^^xsd:float`` are; it is never read as an individual called ``1.5f``.
|
|
63
|
+
What is not a literal form stays an individual name: ``1e5`` (no ``f``),
|
|
64
|
+
``1.5d``, ``1.5fx``;
|
|
65
|
+
* the data ranges: a datatype name, ``DR and DR``, ``DR or DR``, ``not DR``,
|
|
66
|
+
``{lit, lit}`` and the facet bracket ``xsd:decimal[>= 10000, <= 30000]``
|
|
67
|
+
(``>=``/``<=``/``>``/``<`` for the four ordering facets, with a bare numeral
|
|
68
|
+
typed as the base datatype). Read on its own by
|
|
69
|
+
:func:`parse_manchester_data_range`.
|
|
70
|
+
|
|
71
|
+
Manchester Syntax has no declaration table, so a restriction's property is not
|
|
72
|
+
known to be a data property. The parser needs ONE signal, and the FILLER gives
|
|
73
|
+
it, because OWL 2 forbids a name being both a class and a datatype: a filler is
|
|
74
|
+
a data range when it names a built-in datatype of the OWL 2 datatype map
|
|
75
|
+
(``xsd:integer``, ``rdfs:Literal``, …, either spelling), when it is followed by
|
|
76
|
+
a facet bracket, when it is a brace-enclosed list of literals, or when its name
|
|
77
|
+
is one of the ``datatypes=`` the caller passes. A datatype DEFINED by the
|
|
78
|
+
ontology (``OboRoIdrange1``) has no marker a context-free parse could see:
|
|
79
|
+
``d some OboRoIdrange1`` reads as an object restriction over a class of that
|
|
80
|
+
name unless ``datatypes=("OboRoIdrange1",)`` says otherwise. ``hasPet some Dog``
|
|
81
|
+
is unchanged. A built-in datatype name in CLASS position is refused by name
|
|
82
|
+
(it cannot be a class), and an unqualified ``d min 2`` has no filler to tell
|
|
83
|
+
by and stays an object restriction.
|
|
84
|
+
|
|
85
|
+
Qualified number restrictions
|
|
86
|
+
------------------------------
|
|
87
|
+
``min``/``max``/``exactly`` parse into
|
|
88
|
+
:class:`~unicode_logic_kit.dl.concepts.AtLeast` /
|
|
89
|
+
:class:`~unicode_logic_kit.dl.concepts.AtMost` / their conjunction, per the W3C
|
|
90
|
+
grammar's ``objectPropertyExpression ('min'|'max'|'exactly') nonNegativeInteger
|
|
91
|
+
primary?`` production (`§2.2 <https://www.w3.org/TR/owl2-manchester-syntax/#Class_Expressions>`_):
|
|
92
|
+
the qualifying class is a ``primary`` slot exactly like ``some``/``only``'s
|
|
93
|
+
operand (so it parenthesises/binds by the same precedence rule — see "Grammar
|
|
94
|
+
and precedence" below), but it is OPTIONAL, defaulting to ``owl:Thing`` (⊤)
|
|
95
|
+
when the token right after the integer cannot start one (i.e. is not ``not``,
|
|
96
|
+
``inverse``, ``{``, ``(``, or a NAME). ``exactly n`` desugars to
|
|
97
|
+
``AtLeast(n, role, C) ⊓ AtMost(n, role, C)`` at PARSE time (there is no
|
|
98
|
+
separate "exactly" AST node — see :mod:`unicode_logic_kit.dl.concepts`), and
|
|
99
|
+
:func:`to_manchester` renders the two restrictions back out separately (as
|
|
100
|
+
``role min n ... and role max n ...``), which still round-trips to the exact
|
|
101
|
+
same conjunction rather than needing to be recognised and re-folded into
|
|
102
|
+
``exactly``. :func:`to_manchester` likewise omits the qualifying-class text
|
|
103
|
+
entirely when it is ⊤ (``role min n``, not ``role min n owl:Thing``), the
|
|
104
|
+
same "unqualified restriction" shorthand real OWL tooling uses; both
|
|
105
|
+
spellings parse back to the identical ``AtLeast``/``AtMost`` with ``Top()``
|
|
106
|
+
as the filler. See :mod:`unicode_logic_kit.dl.tableau`'s "Qualified number
|
|
107
|
+
restrictions" section for the reasoning side (the *simple roles* restriction
|
|
108
|
+
in particular: a number restriction is rejected outright, by role name, if
|
|
109
|
+
the role is transitive or has a transitive sub-role).
|
|
110
|
+
|
|
111
|
+
Grammar and precedence
|
|
112
|
+
-----------------------
|
|
113
|
+
The W3C grammar (`§2.2 <https://www.w3.org/TR/owl2-manchester-syntax/#Class_Expressions>`_)
|
|
114
|
+
is, in its own words, ambiguous "as stated" and resolved by later productions
|
|
115
|
+
binding *more tightly*::
|
|
116
|
+
|
|
117
|
+
description ::= conjunction ('or' conjunction)* -- loosest
|
|
118
|
+
conjunction ::= primary ('and' primary)*
|
|
119
|
+
primary ::= 'not' primary | restriction | atomic
|
|
120
|
+
restriction ::= NAME 'some' primary | NAME 'only' primary | ...
|
|
121
|
+
atomic ::= NAME | '(' description ')' -- tightest
|
|
122
|
+
|
|
123
|
+
i.e. restrictions (``some``/``only``) bind tightest, then ``not``, then
|
|
124
|
+
``and``, then ``or`` loosest — precisely the precedence lattice
|
|
125
|
+
:mod:`unicode_logic_kit.dl.concepts` already uses for the glyph syntax
|
|
126
|
+
(``_PREC``: ``Or=1 < And=2 < Not=Exists=ForAll=3 < Atomic=4``), so
|
|
127
|
+
:func:`to_manchester` reuses that exact lattice, just spelling the operators
|
|
128
|
+
as keywords instead of glyphs. Two consequences worth spelling out because
|
|
129
|
+
they are easy to get backwards:
|
|
130
|
+
|
|
131
|
+
- ``not`` binds *weaker* than ``some``/``only``: ``not r some A`` parses as
|
|
132
|
+
``not (r some A)`` (¬∃r.A) — ``primary``'s ``'not' primary`` production
|
|
133
|
+
recurses into a ``primary`` that can itself *be* the whole restriction, so
|
|
134
|
+
``not`` scopes over it, not over the role name alone (a bare role has no
|
|
135
|
+
negation in ALC in the first place).
|
|
136
|
+
- ``and`` binds tighter than ``or``: ``A and r some B or C`` parses as
|
|
137
|
+
``(A and (r some B)) or C`` — first ``and`` groups ``A`` with the
|
|
138
|
+
restriction ``r some B`` (restrictions bind tighter than ``and``), then
|
|
139
|
+
the result is ``or``-ed with ``C`` at the loosest level.
|
|
140
|
+
- ``and``/``or`` chains are flat in the grammar (``primary ('and' primary)*``)
|
|
141
|
+
but the AST is binary, so a chain of three or more folds **left**:
|
|
142
|
+
``A and B and C`` is ``(A and B) and C``, matching
|
|
143
|
+
:mod:`unicode_logic_kit.dl.parser`'s glyph parser exactly. So :func:`to_manchester`
|
|
144
|
+
writes a nested operand of the SAME connective without parentheses on the
|
|
145
|
+
left only: ``And(A, And(B, C))`` is ``A and (B and C)``, never ``A and B and
|
|
146
|
+
C``, which reads back as the other tree.
|
|
147
|
+
|
|
148
|
+
``and``/``or``/``not``/``some``/``only``/``min``/``max``/``exactly``/
|
|
149
|
+
``value``/``inverse`` are case-sensitively lowercase keywords; ``Self`` is
|
|
150
|
+
capitalised; ``SubClassOf``/``EquivalentTo`` (used only by
|
|
151
|
+
:func:`parse_manchester_axiom`) are capitalised, with or without a trailing
|
|
152
|
+
colon (``SubClassOf`` and ``SubClassOf:`` are accepted identically — the W3C
|
|
153
|
+
grammar's frame header is colon-terminated, ``SubClassOf:``, but the
|
|
154
|
+
colon-free spelling is the common informal form for a standalone axiom and
|
|
155
|
+
is what callers of this module are expected to write). Per the W3C grammar
|
|
156
|
+
("Prefixes in abbreviated IRIs must not match any of the keywords of this
|
|
157
|
+
syntax"), none of these words may be used as a bare class or role name.
|
|
158
|
+
|
|
159
|
+
Top and Bottom
|
|
160
|
+
--------------
|
|
161
|
+
The Manchester grammar has no dedicated ⊤/⊥ keyword: ``owl:Thing`` and
|
|
162
|
+
``owl:Nothing`` are ordinary ``classIRI``s (the prefix ``owl:`` abbreviating
|
|
163
|
+
``http://www.w3.org/2002/07/owl#``) that merely happen to name the universal
|
|
164
|
+
and empty OWL classes. Since :mod:`unicode_logic_kit.dl.concepts` *does* carry
|
|
165
|
+
first-class :class:`~unicode_logic_kit.dl.concepts.Top`/:class:`~unicode_logic_kit.dl.concepts.Bottom`
|
|
166
|
+
constructors, this module special-cases exactly those two names, in either
|
|
167
|
+
spelling (``owl:Thing`` or the full IRI, bracketed or not; nothing else under the
|
|
168
|
+
``owl:`` prefix is recognised) rather than leaving them as
|
|
169
|
+
opaque :class:`~unicode_logic_kit.dl.concepts.Atomic` names, so
|
|
170
|
+
``owl:Thing and C`` reasons exactly like ``⊤ ⊓ C`` under
|
|
171
|
+
:func:`unicode_logic_kit.dl.tableau.concept_satisfiable` and friends.
|
|
172
|
+
|
|
173
|
+
Full IRIs are one name
|
|
174
|
+
----------------------
|
|
175
|
+
A name in angle brackets with a scheme, ``<http://x.org/a,b>``, is ONE atomic
|
|
176
|
+
name whatever it contains: ``,``, parentheses and square brackets are legal
|
|
177
|
+
inside an IRI, so the tokenizer reads the whole ``<...>`` (and a ``"..."^^<IRI>``
|
|
178
|
+
literal's datatype) before it looks for structural characters. The name is
|
|
179
|
+
STORED WITHOUT its brackets (``Atomic("http://x.org/a,b")``), in every position
|
|
180
|
+
-- a class, a datatype, an object or data property, an individual, a literal's
|
|
181
|
+
datatype, a facet's base -- which is what
|
|
182
|
+
:func:`~unicode_logic_kit.dl.owl_functional.parse_owl_functional` stores too, so
|
|
183
|
+
one IRI is ONE Python string whichever reader produced it: a datatype read here
|
|
184
|
+
matches the literals of that datatype, and ``owl:Thing`` is ``owl:Thing``
|
|
185
|
+
whether it is written ``owl:Thing`` or ``<http://www.w3.org/2002/07/owl#Thing>``
|
|
186
|
+
(the reserved names are compared in their canonical spelling, in every position
|
|
187
|
+
where one of them is special or refused -- a class, a data range, and the
|
|
188
|
+
datatype of a literal). The writers bracket on the way out, and only there:
|
|
189
|
+
:func:`to_manchester` writes a name that is a full IRI (it holds ``://``), or
|
|
190
|
+
that holds a structural character a bare name could not carry, as ``<name>``,
|
|
191
|
+
which reads back as the same string. What tells an IRI from the facet symbols
|
|
192
|
+
``<`` and ``<=`` is its scheme letter: a facet bound starts with a digit, a sign,
|
|
193
|
+
``=``, a dot or a quote, never a letter.
|
|
194
|
+
|
|
195
|
+
Names with no spelling
|
|
196
|
+
-----------------------
|
|
197
|
+
A name is written so that the reader reads back THAT name, or it is refused: the
|
|
198
|
+
writer asks the reader's own tokenizer (and, for a class and an individual, its
|
|
199
|
+
parser) whether the spelling reads back as exactly that name, and raises
|
|
200
|
+
:class:`ValueError` naming the name and the reason when no spelling does. The
|
|
201
|
+
bracketed form is the only escape the syntax has, and it needs a scheme and no
|
|
202
|
+
whitespace, so these have none:
|
|
203
|
+
|
|
204
|
+
* a keyword (``and``, ``some``, ``Domain``, ``SubClassOf:``, ...): ``<and>`` has
|
|
205
|
+
no scheme and reads as the name ``<and>``;
|
|
206
|
+
* a name with whitespace, and the empty name (``A and B`` is a conjunction);
|
|
207
|
+
* a name with a structural character ``( ) { } [ ] ,`` and no scheme;
|
|
208
|
+
* a name that already holds the brackets of a full IRI (``<http://x.org/a>``):
|
|
209
|
+
the reader stores an IRI without them, so the text reads back as another name;
|
|
210
|
+
* ``owl:Thing`` and ``owl:Nothing`` as a CLASS, bracketed or not, in either
|
|
211
|
+
spelling: they are the top and the bottom class (as a role or an individual they
|
|
212
|
+
are ordinary names);
|
|
213
|
+
* a built-in datatype (``xsd:integer``, ``rdfs:Literal``, ``owl:real``, ..., in
|
|
214
|
+
either spelling of its namespace, bracketed or not) as a CLASS: the reader
|
|
215
|
+
tells a data restriction from an object restriction by its filler, so
|
|
216
|
+
``r some xsd:integer`` is a DATA restriction whatever the writer meant, and
|
|
217
|
+
the angle brackets of a full IRI change nothing. As a role, an individual or a
|
|
218
|
+
datatype the name is an ordinary one.
|
|
219
|
+
|
|
220
|
+
A text the reader refuses, as opposed to reads as something else, is not a
|
|
221
|
+
reason to refuse the name.
|
|
222
|
+
|
|
223
|
+
Export-only asymmetry: inverse roles and nominals
|
|
224
|
+
-----------------------------------------------------
|
|
225
|
+
:func:`to_manchester` also RENDERS :class:`~unicode_logic_kit.dl.concepts.InverseRole`
|
|
226
|
+
(as ``inverse r``, in the ``role`` slot of ``some``/``only``/``min``/``max``)
|
|
227
|
+
and :class:`~unicode_logic_kit.dl.concepts.Nominal` (as ``{a}``) — real,
|
|
228
|
+
W3C-legal Manchester syntax the grammar always had room for (see "Rejected
|
|
229
|
+
constructs" above). :func:`parse_manchester` does NOT gain the matching
|
|
230
|
+
read side: ``INVERSE``/``LBRACE`` still hit the same ``_reject`` calls they
|
|
231
|
+
always did, unchanged. This is deliberate, not an oversight: rendering has
|
|
232
|
+
no soundness consequence, so it costs nothing to let a concept built with
|
|
233
|
+
either construct (e.g. from :mod:`unicode_logic_kit.dl.owl_reasoner`'s ALCHQ +
|
|
234
|
+
I/O fragment) still be printed/diffed/logged in this syntax, while parsing it
|
|
235
|
+
back in would re-admit exactly the constructs this module's ALC(HQ)-only
|
|
236
|
+
fragment exists to keep out. ``to_manchester(c)`` followed by
|
|
237
|
+
``parse_manchester`` on the result therefore round-trips for an
|
|
238
|
+
:class:`~unicode_logic_kit.dl.concepts.InverseRole`/:class:`~unicode_logic_kit.dl.concepts.Nominal`-free
|
|
239
|
+
``c`` exactly as before, and raises :class:`ManchesterSyntaxError` — naming
|
|
240
|
+
the construct, as always — for one that uses either.
|
|
241
|
+
|
|
242
|
+
Round-trip guarantee
|
|
243
|
+
---------------------
|
|
244
|
+
``parse_manchester(to_manchester(c)) == c`` for every concept ``c`` the reader
|
|
245
|
+
reads — every constructor but :class:`~unicode_logic_kit.dl.concepts.Nominal` and
|
|
246
|
+
an :class:`~unicode_logic_kit.dl.concepts.InverseRole` role, which are printed and
|
|
247
|
+
refused by name on the way back — see ``tests/test_owl_manchester.py`` for the
|
|
248
|
+
hand-checked precedence cases and ``tests/test_owl_manchester_roundtrip.py`` for
|
|
249
|
+
every constructor in every child slot and for random nestings. The
|
|
250
|
+
grammar/lattice argument above is why this holds structurally, not just on the
|
|
251
|
+
tested examples: :func:`to_manchester` parenthesises a child exactly when its
|
|
252
|
+
precedence is below the parent slot's threshold — the right operand of ``and``
|
|
253
|
+
(``or``) counting a nested ``and`` (``or``) as below it, because the reader folds
|
|
254
|
+
a flat chain to the left — the same rule :func:`parse_manchester` uses to
|
|
255
|
+
*resolve* precedence when reading text back in.
|
|
256
|
+
|
|
257
|
+
Role axioms: the one-line role-box frames
|
|
258
|
+
--------------------------------------------
|
|
259
|
+
:func:`parse_manchester_role_axiom` reads the role-box axiom shapes
|
|
260
|
+
:class:`~unicode_logic_kit.dl.tableau.TBox` holds — see "Role hierarchies and
|
|
261
|
+
transitive roles (RBox)" and "The rest of the OWL 2 role box" in
|
|
262
|
+
:mod:`unicode_logic_kit.dl.tableau`'s module docstring — each as the one-line
|
|
263
|
+
spelling this module already uses for :func:`parse_manchester_axiom`, rather
|
|
264
|
+
than the full multi-line W3C ``ObjectProperty:`` frame (whose annotation
|
|
265
|
+
slots this kit's role box does not represent). Seven shapes, every frame
|
|
266
|
+
keyword accepted with or without its W3C-grammar trailing colon exactly
|
|
267
|
+
like ``SubClassOf``/``SubClassOf:`` above:
|
|
268
|
+
|
|
269
|
+
* ``"r SubPropertyOf s"`` -> ``("subproperty", "r", "s")``
|
|
270
|
+
* ``"r EquivalentTo s"`` -> ``("equivalentproperty", "r", "s")``
|
|
271
|
+
* ``"r InverseOf s"`` -> ``("inverse", "r", "s")``
|
|
272
|
+
* ``"r DisjointWith s"`` -> ``("disjoint", "r", "s")``
|
|
273
|
+
* ``"r Domain: C"`` -> ``("domain", "r", Concept)``
|
|
274
|
+
* ``"r Range: C"`` -> ``("range", "r", Concept)``
|
|
275
|
+
* ``"r Characteristics: X"`` -> ``("transitive"/"symmetric"/…, "r")``, for
|
|
276
|
+
all SEVEN of OWL 2's object-property characteristics (``Transitive``,
|
|
277
|
+
``Symmetric``, ``Asymmetric``, ``Reflexive``, ``Irreflexive``,
|
|
278
|
+
``Functional``, ``InverseFunctional``).
|
|
279
|
+
|
|
280
|
+
``Domain:``/``Range:`` are the two whose right-hand side is a CLASS
|
|
281
|
+
EXPRESSION rather than a role name, parsed by the same ``_description()``
|
|
282
|
+
every other class-expression position uses (so ``"r Domain: A and B"``
|
|
283
|
+
reads), which is why :func:`parse_manchester_role_axiom`'s return type is
|
|
284
|
+
``Tuple[object, ...]`` and not ``Tuple[str, ...]``.
|
|
285
|
+
|
|
286
|
+
A PARSER never refuses an axiom KIND — a ``TBox`` is what a parser fills from
|
|
287
|
+
a file, and which kinds the in-house tableau decides is recorded in
|
|
288
|
+
``dl.tableau._AXIOM_KINDS`` and enforced at query time (see "The axiom-kind
|
|
289
|
+
table" there). So all seven characteristics are read here, and three of them
|
|
290
|
+
(``Symmetric``, ``Reflexive``, ``InverseFunctional``) then make the TABLEAU
|
|
291
|
+
refuse the knowledge base by name while the FOL image still renders them. An
|
|
292
|
+
UNKNOWN characteristic word is still a syntax error naming itself.
|
|
293
|
+
|
|
294
|
+
Two shapes stay refused by name, and the refusal names
|
|
295
|
+
:func:`~unicode_logic_kit.dl.owl_functional.parse_owl_functional` as the entry
|
|
296
|
+
point that does read them:
|
|
297
|
+
|
|
298
|
+
* a PROPERTY CHAIN (``"r o s SubPropertyOf t"``). The bare name ``o`` is
|
|
299
|
+
deliberately NOT promoted to a keyword, so a class or role literally named
|
|
300
|
+
``o`` keeps working in :func:`parse_manchester`;
|
|
301
|
+
* the comma-separated n-ary ``DisjointWith``/``EquivalentTo`` frame slot.
|
|
302
|
+
``DisjointWith`` in particular needs ALL pairs, and a one-axiom parser
|
|
303
|
+
returning one pair would quietly produce a weaker theory.
|
|
304
|
+
|
|
305
|
+
One deliberate asymmetry with ``parse_owl_functional``, documented in both: an
|
|
306
|
+
OWL 2 built-in property name (``owl:topObjectProperty`` and the other three)
|
|
307
|
+
is refused here in EVERY position, including the tautological super-role case
|
|
308
|
+
that ``parse_owl_functional`` consumes as a no-op — a single-axiom parser has
|
|
309
|
+
no return shape for "this axiom is nothing", so it names the entry point that
|
|
310
|
+
does. :func:`role_axiom_to_manchester` is the dual renderer for every shape
|
|
311
|
+
above.
|
|
312
|
+
"""
|
|
313
|
+
|
|
314
|
+
import functools
|
|
315
|
+
import re
|
|
316
|
+
from typing import FrozenSet, Iterable, List, Optional, Tuple
|
|
317
|
+
|
|
318
|
+
from .concepts import (
|
|
319
|
+
Concept, Top, Bottom, Atomic, Not, And, Or, Exists, ForAll, AtLeast, AtMost,
|
|
320
|
+
InverseRole, Nominal, HasValue, DataExists, DataForAll, DataHasValue,
|
|
321
|
+
DataAtLeast, DataAtMost,
|
|
322
|
+
)
|
|
323
|
+
from .datatypes import (
|
|
324
|
+
BUILTIN_DATATYPES, DataRange, Datatype, DatatypeRestriction, DataOneOf,
|
|
325
|
+
DataComplementOf, DataIntersectionOf, DataUnionOf, Literal,
|
|
326
|
+
UnsupportedDatatypeError, canonical_datatype_name, render_literal_fs,
|
|
327
|
+
EXACT_NUMBER_DATATYPES as _EXACT_BASES,
|
|
328
|
+
)
|
|
329
|
+
from .tableau import (
|
|
330
|
+
RESERVED_TOP_ROLES, RoleExpressionError, _reject_concept_role,
|
|
331
|
+
_reject_concept_roles_deep, reserved_role,
|
|
332
|
+
)
|
|
333
|
+
|
|
334
|
+
__all__ = [
|
|
335
|
+
"parse_manchester", "to_manchester", "parse_manchester_axiom",
|
|
336
|
+
"parse_manchester_role_axiom", "role_axiom_to_manchester",
|
|
337
|
+
"parse_manchester_data_range", "to_manchester_data_range",
|
|
338
|
+
"parse_manchester_literal", "ManchesterSyntaxError",
|
|
339
|
+
]
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
class ManchesterSyntaxError(ValueError):
|
|
343
|
+
"""Raised by this module's parsers on malformed input *and* on syntax
|
|
344
|
+
that is valid Manchester/OWL 2 but falls outside the ALC fragment (see
|
|
345
|
+
the module docstring's "Rejected constructs"). Always a :class:`ValueError`
|
|
346
|
+
subclass with a message naming the specific offending construct and its
|
|
347
|
+
position in the input, per the kit's honesty convention.
|
|
348
|
+
"""
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
# --------------------------------------------------------------------------- #
|
|
352
|
+
# Tokenizer.
|
|
353
|
+
# --------------------------------------------------------------------------- #
|
|
354
|
+
|
|
355
|
+
# Lowercase connective/restriction keywords (case-sensitive, per the W3C grammar).
|
|
356
|
+
_KEYWORDS = {
|
|
357
|
+
"and": "AND", "or": "OR", "not": "NOT", "some": "SOME", "only": "ONLY",
|
|
358
|
+
"min": "MIN", "max": "MAX", "exactly": "EXACTLY", "value": "VALUE",
|
|
359
|
+
"Self": "SELF", "inverse": "INVERSE",
|
|
360
|
+
}
|
|
361
|
+
|
|
362
|
+
# Axiom-frame keywords, accepted with or without a trailing colon (see module docstring).
|
|
363
|
+
_AXIOM_KEYWORDS = {"SubClassOf": "SUBCLASSOF", "EquivalentTo": "EQUIVALENTTO"}
|
|
364
|
+
|
|
365
|
+
# Role-axiom-frame keywords (see "Role axioms" in the module docstring), same
|
|
366
|
+
# colon-optional convention as _AXIOM_KEYWORDS above. `EquivalentTo` is NOT
|
|
367
|
+
# here: it is already an _AXIOM_KEYWORDS entry (shared with the class-axiom
|
|
368
|
+
# frame) and _classify_word checks that table first, so the role-axiom parser
|
|
369
|
+
# accepts its EQUIVALENTTO token directly.
|
|
370
|
+
_ROLE_AXIOM_KEYWORDS = {
|
|
371
|
+
"SubPropertyOf": "SUBPROPERTYOF",
|
|
372
|
+
"Characteristics": "CHARACTERISTICS",
|
|
373
|
+
"InverseOf": "INVERSEOF",
|
|
374
|
+
"DisjointWith": "DISJOINTWITH",
|
|
375
|
+
"Domain": "DOMAIN",
|
|
376
|
+
"Range": "RANGE",
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
_STRUCT_TOKENS = {
|
|
380
|
+
"(": "LPAREN", ")": "RPAREN",
|
|
381
|
+
"{": "LBRACE", "}": "RBRACE",
|
|
382
|
+
"[": "LBRACKET", "]": "RBRACKET",
|
|
383
|
+
",": "COMMA",
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
# A Token is (type: str, value: str, pos: int).
|
|
387
|
+
_Token = Tuple[str, str, int]
|
|
388
|
+
|
|
389
|
+
#: A full IRI in angle brackets -- ``<http://x.org/a,b>`` -- which is ONE atomic
|
|
390
|
+
#: name whatever it contains (``,``, parentheses and brackets are all legal inside
|
|
391
|
+
#: an IRI), so the tokenizer reads it before it looks for structural characters.
|
|
392
|
+
#: It must open with a scheme (``letter (letter|digit|+|.|-)* ':'``) and contain
|
|
393
|
+
#: no whitespace; that is what tells it from the facet symbols ``<`` and ``<=``
|
|
394
|
+
#: (``xsd:integer[<5,>=3]``), whose next character is a digit, ``=``, a sign, a
|
|
395
|
+
#: dot or a quote -- never a letter.
|
|
396
|
+
_FULL_IRI = re.compile(r"<[A-Za-z][A-Za-z0-9+.\-]*:[^\s>]*>")
|
|
397
|
+
|
|
398
|
+
|
|
399
|
+
def _unbracket(name: str) -> str:
|
|
400
|
+
"""``name`` without the angle brackets of a full IRI: the ONE stored spelling
|
|
401
|
+
of an IRI (see "Full IRIs are one name" in the module docstring). Any other
|
|
402
|
+
name is returned unchanged."""
|
|
403
|
+
return name[1:-1] if _FULL_IRI.fullmatch(name) else name
|
|
404
|
+
|
|
405
|
+
|
|
406
|
+
def _classify_word(word: str) -> str:
|
|
407
|
+
"""Classify a maximal non-structural, non-whitespace run as a keyword or NAME."""
|
|
408
|
+
if word in _KEYWORDS:
|
|
409
|
+
return _KEYWORDS[word]
|
|
410
|
+
bare = word[:-1] if word.endswith(":") else word
|
|
411
|
+
if bare in _AXIOM_KEYWORDS:
|
|
412
|
+
return _AXIOM_KEYWORDS[bare]
|
|
413
|
+
if bare in _ROLE_AXIOM_KEYWORDS:
|
|
414
|
+
return _ROLE_AXIOM_KEYWORDS[bare]
|
|
415
|
+
return "NAME"
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def _tokenize(text: str) -> List[_Token]:
|
|
419
|
+
"""Split ``text`` into structural/keyword/NAME tokens plus a trailing EOF
|
|
420
|
+
sentinel (whose ``pos`` is ``len(text)``, for error messages).
|
|
421
|
+
"""
|
|
422
|
+
tokens: List[_Token] = []
|
|
423
|
+
i, n = 0, len(text)
|
|
424
|
+
while i < n:
|
|
425
|
+
ch = text[i]
|
|
426
|
+
if ch.isspace():
|
|
427
|
+
i += 1
|
|
428
|
+
continue
|
|
429
|
+
if ch in _STRUCT_TOKENS:
|
|
430
|
+
tokens.append((_STRUCT_TOKENS[ch], ch, i))
|
|
431
|
+
i += 1
|
|
432
|
+
continue
|
|
433
|
+
if ch == "<":
|
|
434
|
+
iri = _FULL_IRI.match(text, i)
|
|
435
|
+
# ... and it must END there: `<http://x.org/a>b` is one odd word, as
|
|
436
|
+
# it always was, not an IRI followed by a name.
|
|
437
|
+
if iri is not None and (iri.end() == n or text[iri.end()].isspace()
|
|
438
|
+
or text[iri.end()] in _STRUCT_TOKENS):
|
|
439
|
+
tokens.append(("NAME", iri.group(0)[1:-1], i)) # stored WITHOUT brackets
|
|
440
|
+
i = iri.end()
|
|
441
|
+
continue
|
|
442
|
+
if ch == '"':
|
|
443
|
+
# A quoted literal is ONE token, with its ^^datatype or @language
|
|
444
|
+
# suffix: its text may hold whitespace and structural characters
|
|
445
|
+
# (the lexical form of a string), which nothing else here may.
|
|
446
|
+
start = i
|
|
447
|
+
i += 1
|
|
448
|
+
while i < n and text[i] != '"':
|
|
449
|
+
i += 2 if text[i] == "\\" and i + 1 < n else 1
|
|
450
|
+
if i >= n:
|
|
451
|
+
raise ManchesterSyntaxError(
|
|
452
|
+
f"parse_manchester: unterminated string literal starting "
|
|
453
|
+
f"at position {start} in {text!r}")
|
|
454
|
+
i += 1
|
|
455
|
+
iri = _FULL_IRI.match(text, i + 2) if text.startswith("^^", i) else None
|
|
456
|
+
if iri is not None:
|
|
457
|
+
i = iri.end() # "..."^^<IRI>: the datatype IRI is one name
|
|
458
|
+
while i < n and not text[i].isspace() and text[i] not in _STRUCT_TOKENS:
|
|
459
|
+
i += 1
|
|
460
|
+
tokens.append(("LITERAL", text[start:i], start))
|
|
461
|
+
continue
|
|
462
|
+
start = i
|
|
463
|
+
while i < n and not text[i].isspace() and text[i] not in _STRUCT_TOKENS:
|
|
464
|
+
i += 1
|
|
465
|
+
word = text[start:i]
|
|
466
|
+
tokens.append((_classify_word(word), word, start))
|
|
467
|
+
tokens.append(("EOF", "", n))
|
|
468
|
+
return tokens
|
|
469
|
+
|
|
470
|
+
|
|
471
|
+
# --------------------------------------------------------------------------- #
|
|
472
|
+
# Recursive-descent parser (description expressions only; parse_manchester_axiom
|
|
473
|
+
# splits an axiom into two token slices and runs one of these per side).
|
|
474
|
+
# --------------------------------------------------------------------------- #
|
|
475
|
+
|
|
476
|
+
class _Parser:
|
|
477
|
+
"""A single parse of one token slice; not re-used across calls."""
|
|
478
|
+
|
|
479
|
+
def __init__(self, tokens: List[_Token], text: str,
|
|
480
|
+
datatypes: Iterable[str] = ()):
|
|
481
|
+
self._tokens = tokens
|
|
482
|
+
self._text = text
|
|
483
|
+
self._i = 0
|
|
484
|
+
self._datatypes: FrozenSet[str] = frozenset(
|
|
485
|
+
canonical_datatype_name(_unbracket(name)) for name in datatypes)
|
|
486
|
+
|
|
487
|
+
def _peek(self) -> _Token:
|
|
488
|
+
return self._tokens[self._i]
|
|
489
|
+
|
|
490
|
+
def _advance(self) -> _Token:
|
|
491
|
+
tok = self._tokens[self._i]
|
|
492
|
+
self._i += 1
|
|
493
|
+
return tok
|
|
494
|
+
|
|
495
|
+
def _error(self, message: str) -> ManchesterSyntaxError:
|
|
496
|
+
return ManchesterSyntaxError(f"{message} in {self._text!r}")
|
|
497
|
+
|
|
498
|
+
def _expect(self, ttype: str, what: str) -> _Token:
|
|
499
|
+
tok = self._peek()
|
|
500
|
+
if tok[0] != ttype:
|
|
501
|
+
found = "end of input" if tok[0] == "EOF" else f"{tok[1]!r}"
|
|
502
|
+
raise self._error(
|
|
503
|
+
f"parse_manchester: expected {what} but found {found} "
|
|
504
|
+
f"at position {tok[2]}")
|
|
505
|
+
return self._advance()
|
|
506
|
+
|
|
507
|
+
def _expect_eof(self) -> None:
|
|
508
|
+
tok = self._peek()
|
|
509
|
+
if tok[0] != "EOF":
|
|
510
|
+
raise self._error(
|
|
511
|
+
f"parse_manchester: unexpected trailing input {tok[1]!r} "
|
|
512
|
+
f"at position {tok[2]}")
|
|
513
|
+
|
|
514
|
+
def _reject(self, what: str, tok: _Token) -> None:
|
|
515
|
+
raise self._error(
|
|
516
|
+
f"parse_manchester: {what} — not supported outside ALC "
|
|
517
|
+
f"(found {tok[1]!r} at position {tok[2]})")
|
|
518
|
+
|
|
519
|
+
def _checked(self, concept: Concept, pos: int) -> Concept:
|
|
520
|
+
"""``concept``, unless its role is an OWL 2 built-in property name (or
|
|
521
|
+
``=``/``≠``) — refused BY NAME, as :func:`parse_manchester_role_axiom`
|
|
522
|
+
and ``dl.parse_owl_functional`` refuse the same name. ``owl:topObject
|
|
523
|
+
Property some A`` used to read as an ordinary role of that name, whose
|
|
524
|
+
verdict is not the universal property's. ``exactly`` is the ``And`` of
|
|
525
|
+
two restrictions over ONE role, so the left one stands for both."""
|
|
526
|
+
try:
|
|
527
|
+
_reject_concept_role(
|
|
528
|
+
concept.left if isinstance(concept, And) else concept,
|
|
529
|
+
where="parse_manchester")
|
|
530
|
+
except RoleExpressionError as exc:
|
|
531
|
+
raise self._error(f"{exc} (at position {pos})") from exc
|
|
532
|
+
return concept
|
|
533
|
+
|
|
534
|
+
# -- grammar levels, loosest first (mirrors the W3C production order) -- #
|
|
535
|
+
|
|
536
|
+
def _description(self) -> Concept:
|
|
537
|
+
left = self._conjunction()
|
|
538
|
+
while self._peek()[0] == "OR":
|
|
539
|
+
self._advance()
|
|
540
|
+
left = Or(left, self._conjunction())
|
|
541
|
+
return left
|
|
542
|
+
|
|
543
|
+
def _conjunction(self) -> Concept:
|
|
544
|
+
left = self._primary()
|
|
545
|
+
while self._peek()[0] == "AND":
|
|
546
|
+
self._advance()
|
|
547
|
+
left = And(left, self._primary())
|
|
548
|
+
return left
|
|
549
|
+
|
|
550
|
+
def _primary(self) -> Concept:
|
|
551
|
+
ttype, value, pos = self._peek()
|
|
552
|
+
if ttype == "NOT":
|
|
553
|
+
self._advance()
|
|
554
|
+
return Not(self._primary())
|
|
555
|
+
if ttype == "INVERSE":
|
|
556
|
+
self._reject("inverse roles ('inverse r')", self._peek())
|
|
557
|
+
if ttype == "LBRACE":
|
|
558
|
+
self._reject("nominal concepts ('{a, b}')", self._peek())
|
|
559
|
+
if ttype == "LPAREN":
|
|
560
|
+
self._advance()
|
|
561
|
+
inner = self._description()
|
|
562
|
+
self._expect("RPAREN", "')'")
|
|
563
|
+
return inner
|
|
564
|
+
if ttype == "NAME":
|
|
565
|
+
self._advance()
|
|
566
|
+
if self._peek()[0] == "LBRACKET":
|
|
567
|
+
raise self._error(
|
|
568
|
+
f"parse_manchester: the facet bracket after {value!r} (at "
|
|
569
|
+
f"position {self._peek()[2]}) makes it a DATA RANGE, which is "
|
|
570
|
+
f"not a class expression. Use it as the filler of a data "
|
|
571
|
+
f"restriction ('d some {value}[...]') or read it on its own "
|
|
572
|
+
f"with parse_manchester_data_range")
|
|
573
|
+
nxt = self._peek()
|
|
574
|
+
if nxt[0] == "SOME":
|
|
575
|
+
self._advance()
|
|
576
|
+
if self._filler_is_data_range():
|
|
577
|
+
return self._checked(DataExists(value, self._data_primary()), pos)
|
|
578
|
+
return self._checked(Exists(value, self._primary()), pos)
|
|
579
|
+
if nxt[0] == "ONLY":
|
|
580
|
+
self._advance()
|
|
581
|
+
if self._filler_is_data_range():
|
|
582
|
+
return self._checked(DataForAll(value, self._data_primary()), pos)
|
|
583
|
+
return self._checked(ForAll(value, self._primary()), pos)
|
|
584
|
+
if nxt[0] in ("MIN", "MAX", "EXACTLY"):
|
|
585
|
+
self._advance()
|
|
586
|
+
return self._checked(self._number_restriction(nxt[0], value), pos)
|
|
587
|
+
if nxt[0] == "VALUE":
|
|
588
|
+
# `r value a` -> HasValue(r, a); `d value "1"^^xsd:integer` (or a
|
|
589
|
+
# bare numeral) -> DataHasValue(d, lit). The keyword was always
|
|
590
|
+
# tokenised (see _KEYWORDS). NOT Exists(value, Nominal(...)):
|
|
591
|
+
# `r some {a}` is the other spelling and stays refused, so the
|
|
592
|
+
# two round-trip apart -- see dl.concepts.HasValue.
|
|
593
|
+
self._advance()
|
|
594
|
+
literal = self._literal_or_none()
|
|
595
|
+
if literal is not None:
|
|
596
|
+
return self._checked(DataHasValue(value, literal), pos)
|
|
597
|
+
return self._checked(HasValue(
|
|
598
|
+
value, self._expect("NAME", "an individual name or a literal")[1]), pos)
|
|
599
|
+
if nxt[0] == "SELF":
|
|
600
|
+
self._reject("Self restrictions ('Self')", nxt)
|
|
601
|
+
canonical = canonical_datatype_name(value)
|
|
602
|
+
if canonical == "owl:Thing":
|
|
603
|
+
return Top()
|
|
604
|
+
if canonical == "owl:Nothing":
|
|
605
|
+
return Bottom()
|
|
606
|
+
if self._is_datatype(value):
|
|
607
|
+
raise self._error(
|
|
608
|
+
f"parse_manchester: {value!r} (at position {pos}) is a "
|
|
609
|
+
f"DATATYPE, not a class — OWL 2 does not let one name be "
|
|
610
|
+
f"both. Use it as the filler of a data restriction "
|
|
611
|
+
f"('d some {value}') or read it with parse_manchester_data_range")
|
|
612
|
+
return Atomic(value)
|
|
613
|
+
found = "end of input" if ttype == "EOF" else f"{value!r}"
|
|
614
|
+
raise self._error(
|
|
615
|
+
f"parse_manchester: unexpected {found} at position {pos}; "
|
|
616
|
+
"expected a class name, 'not', 'owl:Thing', 'owl:Nothing', or '('")
|
|
617
|
+
|
|
618
|
+
# -- qualified number restrictions ('min'/'max'/'exactly') ---------- #
|
|
619
|
+
|
|
620
|
+
_PRIMARY_START = {"NOT", "INVERSE", "LBRACE", "LPAREN", "NAME"}
|
|
621
|
+
|
|
622
|
+
def _number_restriction(self, kind: str, role: str) -> Concept:
|
|
623
|
+
"""Parse the ``n primary?`` tail of ``role (min|max|exactly) n primary?``,
|
|
624
|
+
having already consumed ``role`` and the ``kind`` keyword (see the module
|
|
625
|
+
docstring's "Qualified number restrictions"). The qualifying class defaults
|
|
626
|
+
to ``owl:Thing`` (⊤) when the next token cannot start a ``primary``.
|
|
627
|
+
"""
|
|
628
|
+
n = self._expect_nonneg_int()
|
|
629
|
+
if self._peek()[0] in self._PRIMARY_START and self._filler_is_data_range():
|
|
630
|
+
datarange = self._data_primary()
|
|
631
|
+
if kind == "MIN":
|
|
632
|
+
return DataAtLeast(n, role, datarange)
|
|
633
|
+
if kind == "MAX":
|
|
634
|
+
return DataAtMost(n, role, datarange)
|
|
635
|
+
return And(DataAtLeast(n, role, datarange), DataAtMost(n, role, datarange))
|
|
636
|
+
filler = self._primary() if self._peek()[0] in self._PRIMARY_START else Top()
|
|
637
|
+
if kind == "MIN":
|
|
638
|
+
return AtLeast(n, role, filler)
|
|
639
|
+
if kind == "MAX":
|
|
640
|
+
return AtMost(n, role, filler)
|
|
641
|
+
return And(AtLeast(n, role, filler), AtMost(n, role, filler)) # 'exactly'
|
|
642
|
+
|
|
643
|
+
def _expect_nonneg_int(self) -> int:
|
|
644
|
+
tok = self._peek()
|
|
645
|
+
if tok[0] == "NAME" and tok[1].isdigit():
|
|
646
|
+
self._advance()
|
|
647
|
+
return int(tok[1])
|
|
648
|
+
found = "end of input" if tok[0] == "EOF" else f"{tok[1]!r}"
|
|
649
|
+
raise self._error(
|
|
650
|
+
f"parse_manchester: expected a non-negative integer but found {found} "
|
|
651
|
+
f"at position {tok[2]}")
|
|
652
|
+
|
|
653
|
+
def parse_description(self) -> Concept:
|
|
654
|
+
c = self._description()
|
|
655
|
+
self._expect_eof()
|
|
656
|
+
return c
|
|
657
|
+
|
|
658
|
+
# -- data ranges and literals ---------------------------------------- #
|
|
659
|
+
|
|
660
|
+
def _is_datatype(self, name: str) -> bool:
|
|
661
|
+
"""True iff ``name`` is a datatype for this parse: one of the OWL 2
|
|
662
|
+
datatype map's (either spelling of its namespace) or one the caller
|
|
663
|
+
declared with ``datatypes=``."""
|
|
664
|
+
canonical = canonical_datatype_name(name)
|
|
665
|
+
return canonical in BUILTIN_DATATYPES or canonical in self._datatypes
|
|
666
|
+
|
|
667
|
+
def _filler_is_data_range(self) -> bool:
|
|
668
|
+
"""Does the filler that starts at the cursor read as a DATA range?
|
|
669
|
+
|
|
670
|
+
Looks only at the filler's HEAD, skipping any ``not`` and ``(`` in front
|
|
671
|
+
of it: a name that is a datatype or is followed by a facet bracket, or a
|
|
672
|
+
brace-enclosed list that starts with a literal. This is the one signal a
|
|
673
|
+
context-free Manchester parse has (see the module docstring's "Data
|
|
674
|
+
restrictions"); it never consumes anything.
|
|
675
|
+
"""
|
|
676
|
+
tokens, i = self._tokens, self._i
|
|
677
|
+
while tokens[i][0] in ("NOT", "LPAREN"):
|
|
678
|
+
i += 1
|
|
679
|
+
ttype, value, _pos = tokens[i]
|
|
680
|
+
if ttype == "NAME":
|
|
681
|
+
return tokens[i + 1][0] == "LBRACKET" or self._is_datatype(value)
|
|
682
|
+
if ttype == "LBRACE":
|
|
683
|
+
head = tokens[i + 1]
|
|
684
|
+
return head[0] == "LITERAL" or (head[0] == "NAME" and _bare_numeral(head[1]))
|
|
685
|
+
return False
|
|
686
|
+
|
|
687
|
+
def _literal_or_none(self) -> Optional[Literal]:
|
|
688
|
+
"""Consume a literal if one is at the cursor: a quoted literal token or
|
|
689
|
+
a bare numeral (``1``, ``-3``, ``1.5``, or the floating-point ``1.5f``);
|
|
690
|
+
else ``None``."""
|
|
691
|
+
ttype, value, pos = self._peek()
|
|
692
|
+
if ttype == "LITERAL":
|
|
693
|
+
self._advance()
|
|
694
|
+
return self._build(lambda: _literal_from_token(value, pos), value, pos)
|
|
695
|
+
if ttype == "NAME" and _bare_numeral(value):
|
|
696
|
+
self._advance()
|
|
697
|
+
return self._build(lambda: _bare_literal(value, None), value, pos)
|
|
698
|
+
return None
|
|
699
|
+
|
|
700
|
+
def _build(self, make, what: str, pos: int):
|
|
701
|
+
"""``make()``, a data-layer refusal re-raised as the parser's own error
|
|
702
|
+
naming ``what`` and where."""
|
|
703
|
+
try:
|
|
704
|
+
return make()
|
|
705
|
+
except UnsupportedDatatypeError as exc:
|
|
706
|
+
raise self._error(f"parse_manchester: {exc} (at position {pos}, {what!r})") from exc
|
|
707
|
+
|
|
708
|
+
def _data_description(self) -> DataRange:
|
|
709
|
+
"""``dataConjunction ('or' dataConjunction)*`` -- an N-ARY union, so a
|
|
710
|
+
chain is one :class:`DataUnionOf`; a parenthesised group stays nested."""
|
|
711
|
+
parts = [self._data_conjunction()]
|
|
712
|
+
while self._peek()[0] == "OR":
|
|
713
|
+
self._advance()
|
|
714
|
+
parts.append(self._data_conjunction())
|
|
715
|
+
return parts[0] if len(parts) == 1 else DataUnionOf(tuple(parts))
|
|
716
|
+
|
|
717
|
+
def _data_conjunction(self) -> DataRange:
|
|
718
|
+
parts = [self._data_primary()]
|
|
719
|
+
while self._peek()[0] == "AND":
|
|
720
|
+
self._advance()
|
|
721
|
+
parts.append(self._data_primary())
|
|
722
|
+
return parts[0] if len(parts) == 1 else DataIntersectionOf(tuple(parts))
|
|
723
|
+
|
|
724
|
+
def _data_primary(self) -> DataRange:
|
|
725
|
+
"""``'not'? dataAtomic`` with ``dataAtomic`` a datatype, a facet
|
|
726
|
+
restriction ``dt[facet lit, …]``, a literal set ``{lit, …}`` or a
|
|
727
|
+
parenthesised data range."""
|
|
728
|
+
ttype, value, pos = self._peek()
|
|
729
|
+
if ttype == "NOT":
|
|
730
|
+
self._advance()
|
|
731
|
+
return DataComplementOf(self._data_primary())
|
|
732
|
+
if ttype == "LPAREN":
|
|
733
|
+
self._advance()
|
|
734
|
+
inner = self._data_description()
|
|
735
|
+
self._expect("RPAREN", "')'")
|
|
736
|
+
return inner
|
|
737
|
+
if ttype == "LBRACE":
|
|
738
|
+
self._advance()
|
|
739
|
+
values = [self._expect_literal()]
|
|
740
|
+
while self._peek()[0] == "COMMA":
|
|
741
|
+
self._advance()
|
|
742
|
+
values.append(self._expect_literal())
|
|
743
|
+
self._expect("RBRACE", "'}'")
|
|
744
|
+
return DataOneOf(tuple(values))
|
|
745
|
+
if ttype == "NAME":
|
|
746
|
+
if canonical_datatype_name(value) in ("owl:Thing", "owl:Nothing"):
|
|
747
|
+
# The mirror of the "is a DATATYPE, not a class" refusal in
|
|
748
|
+
# _primary, and the same policy dl.parse_owl_functional has: a
|
|
749
|
+
# class is not a data range, and a datatype called owl:Thing
|
|
750
|
+
# would be an uninterpreted predicate of that name.
|
|
751
|
+
raise self._error(
|
|
752
|
+
f"parse_manchester: {value!r} (at position {pos}) is a CLASS, "
|
|
753
|
+
f"not a data range — OWL 2 does not let one name be both. "
|
|
754
|
+
f"The data domain's own 'everything' is rdfs:Literal; use a "
|
|
755
|
+
f"datatype name here")
|
|
756
|
+
self._advance()
|
|
757
|
+
if self._peek()[0] == "LBRACKET":
|
|
758
|
+
return self._facet_restriction(value, pos)
|
|
759
|
+
return Datatype(value)
|
|
760
|
+
found = "end of input" if ttype == "EOF" else f"{value!r}"
|
|
761
|
+
raise self._error(
|
|
762
|
+
f"parse_manchester: expected a data range but found {found} at "
|
|
763
|
+
f"position {pos}; expected a datatype name, 'not', '{{', or '('")
|
|
764
|
+
|
|
765
|
+
def _expect_literal(self) -> Literal:
|
|
766
|
+
literal = self._literal_or_none()
|
|
767
|
+
if literal is None:
|
|
768
|
+
found = "end of input" if self._peek()[0] == "EOF" else f"{self._peek()[1]!r}"
|
|
769
|
+
raise self._error(
|
|
770
|
+
f"parse_manchester: expected a literal but found {found} at "
|
|
771
|
+
f"position {self._peek()[2]}")
|
|
772
|
+
return literal
|
|
773
|
+
|
|
774
|
+
def _facet_restriction(self, base: str, pos: int) -> DataRange:
|
|
775
|
+
"""``dt '[' facet literal (',' facet literal)* ']'`` (the name already
|
|
776
|
+
consumed, the cursor on ``[``)."""
|
|
777
|
+
self._expect("LBRACKET", "'['")
|
|
778
|
+
facets: List[Tuple[str, Literal]] = []
|
|
779
|
+
while True:
|
|
780
|
+
ftype, fvalue, fpos = self._peek()
|
|
781
|
+
if ftype != "NAME":
|
|
782
|
+
found = "end of input" if ftype == "EOF" else f"{fvalue!r}"
|
|
783
|
+
raise self._error(
|
|
784
|
+
f"parse_manchester: expected a facet ({', '.join(_FACET_SYMBOLS)}, "
|
|
785
|
+
f"…) but found {found} at position {fpos}")
|
|
786
|
+
self._advance()
|
|
787
|
+
symbol, glued = fvalue, None
|
|
788
|
+
# `>=5` with no space after the symbol: the symbol is the longest
|
|
789
|
+
# ordering prefix, the rest the bound.
|
|
790
|
+
if fvalue not in _FACET_SYMBOLS:
|
|
791
|
+
for candidate in (">=", "<=", ">", "<"):
|
|
792
|
+
if fvalue.startswith(candidate):
|
|
793
|
+
symbol, glued = candidate, fvalue[len(candidate):]
|
|
794
|
+
break
|
|
795
|
+
if symbol not in _FACET_SYMBOLS:
|
|
796
|
+
raise self._error(
|
|
797
|
+
f"parse_manchester: unknown facet {symbol!r} at position "
|
|
798
|
+
f"{fpos}; the facets are {', '.join(_FACET_SYMBOLS)}")
|
|
799
|
+
if glued is not None:
|
|
800
|
+
bound = self._build(lambda: _bare_literal(glued, base), glued, fpos)
|
|
801
|
+
else:
|
|
802
|
+
btype, bvalue, bpos = self._peek()
|
|
803
|
+
if btype == "LITERAL":
|
|
804
|
+
self._advance()
|
|
805
|
+
bound = self._build(lambda: _literal_from_token(bvalue, bpos), bvalue, bpos)
|
|
806
|
+
elif btype == "NAME" and _bare_numeral(bvalue):
|
|
807
|
+
self._advance()
|
|
808
|
+
bound = self._build(lambda: _bare_literal(bvalue, base), bvalue, bpos)
|
|
809
|
+
else:
|
|
810
|
+
found = "end of input" if btype == "EOF" else f"{bvalue!r}"
|
|
811
|
+
raise self._error(
|
|
812
|
+
f"parse_manchester: expected the facet {symbol!r}'s "
|
|
813
|
+
f"literal but found {found} at position {bpos}")
|
|
814
|
+
facets.append((_FACET_SYMBOLS[symbol], bound))
|
|
815
|
+
if self._peek()[0] == "COMMA":
|
|
816
|
+
self._advance()
|
|
817
|
+
continue
|
|
818
|
+
break
|
|
819
|
+
self._expect("RBRACKET", "']'")
|
|
820
|
+
return self._build(
|
|
821
|
+
lambda: DatatypeRestriction(Datatype(base), tuple(facets)), base, pos)
|
|
822
|
+
|
|
823
|
+
def parse_data_range(self) -> DataRange:
|
|
824
|
+
datarange = self._data_description()
|
|
825
|
+
self._expect_eof()
|
|
826
|
+
return datarange
|
|
827
|
+
|
|
828
|
+
|
|
829
|
+
# --------------------------------------------------------------------------- #
|
|
830
|
+
# Literals and facets.
|
|
831
|
+
# --------------------------------------------------------------------------- #
|
|
832
|
+
|
|
833
|
+
#: Manchester Syntax's facet spellings, mapped to the canonical facet names the
|
|
834
|
+
#: data layer uses. The four ordering symbols are the ones this kit translates;
|
|
835
|
+
#: the keyword facets are read so that the REFUSAL can name them.
|
|
836
|
+
_FACET_SYMBOLS = {
|
|
837
|
+
">=": "xsd:minInclusive", "<=": "xsd:maxInclusive",
|
|
838
|
+
">": "xsd:minExclusive", "<": "xsd:maxExclusive",
|
|
839
|
+
"length": "xsd:length", "minLength": "xsd:minLength", "maxLength": "xsd:maxLength",
|
|
840
|
+
"pattern": "xsd:pattern", "langRange": "rdf:langRange",
|
|
841
|
+
"totalDigits": "xsd:totalDigits", "fractionDigits": "xsd:fractionDigits",
|
|
842
|
+
}
|
|
843
|
+
_SYMBOL_OF_FACET = {facet: symbol for symbol, facet in _FACET_SYMBOLS.items()}
|
|
844
|
+
|
|
845
|
+
_BARE_NUMERAL = re.compile(r"[+-]?(?:[0-9]+(?:\.[0-9]+)?|\.[0-9]+)\Z")
|
|
846
|
+
#: The W3C grammar's ``floatingPointLiteral``: a sign, ``digits ['.' digits]`` or
|
|
847
|
+
#: ``'.' digits``, an optional exponent, and the ``f``/``F`` that makes it a
|
|
848
|
+
#: float. Group 1 is the lexical form WITHOUT the suffix (``1.0e+2f`` is the
|
|
849
|
+
#: ``xsd:float`` ``"1.0e+2"``).
|
|
850
|
+
_FLOATING_POINT = re.compile(
|
|
851
|
+
r"([+-]?(?:[0-9]+(?:\.[0-9]+)?|\.[0-9]+)(?:[eE][+-]?[0-9]+)?)[fF]\Z")
|
|
852
|
+
_TYPED_LITERAL = re.compile(r'"((?:[^"\\]|\\.)*)"(?:\^\^(\S+)|@(\S+))?\Z', re.DOTALL)
|
|
853
|
+
|
|
854
|
+
|
|
855
|
+
def _bare_numeral(name: str) -> bool:
|
|
856
|
+
"""Is ``name`` a bare Manchester numeral: an integer (``1``, ``-3``), a
|
|
857
|
+
decimal (``1.5``) or a floating-point literal (``1.5f``, ``1.0e+2F``)?"""
|
|
858
|
+
return (_BARE_NUMERAL.match(name) is not None
|
|
859
|
+
or _FLOATING_POINT.match(name) is not None)
|
|
860
|
+
|
|
861
|
+
|
|
862
|
+
def _bare_literal(text: str, base: Optional[str]) -> Literal:
|
|
863
|
+
"""The literal a BARE numeral stands for: a floating-point literal is an
|
|
864
|
+
``xsd:float`` whatever the facet's base (the ``f`` says so, as the quoted
|
|
865
|
+
``"1.5"^^xsd:float`` does); otherwise typed as the facet's base datatype
|
|
866
|
+
when that is an exact-number datatype (``xsd:decimal[>= 10000]`` bounds are
|
|
867
|
+
decimals), else by its shape — an integer, or a decimal if it has a point."""
|
|
868
|
+
floating = _FLOATING_POINT.match(text)
|
|
869
|
+
if floating is not None:
|
|
870
|
+
return Literal(floating.group(1), "xsd:float")
|
|
871
|
+
canonical = canonical_datatype_name(base) if base is not None else None
|
|
872
|
+
if canonical is not None and canonical in _EXACT_BASES:
|
|
873
|
+
return Literal(text, canonical)
|
|
874
|
+
return Literal(text, "xsd:decimal" if "." in text else "xsd:integer")
|
|
875
|
+
|
|
876
|
+
|
|
877
|
+
def _literal_from_token(token: str, pos: int = 0) -> Literal:
|
|
878
|
+
match = _TYPED_LITERAL.match(token)
|
|
879
|
+
if match is None:
|
|
880
|
+
raise UnsupportedDatatypeError(f"{token!r} is not a literal")
|
|
881
|
+
lexical = re.sub(r'\\(["\\])', r"\1", match.group(1))
|
|
882
|
+
if match.group(3) is not None:
|
|
883
|
+
return Literal(lexical, "xsd:string", match.group(3))
|
|
884
|
+
datatype = match.group(2) or "xsd:string"
|
|
885
|
+
if datatype.startswith("<") and datatype.endswith(">"):
|
|
886
|
+
datatype = datatype[1:-1]
|
|
887
|
+
if canonical_datatype_name(datatype) in ("owl:Thing", "owl:Nothing"):
|
|
888
|
+
# the same refusal a data range gets (see _Parser._data_primary), for the
|
|
889
|
+
# datatype of a literal, in either spelling of the name
|
|
890
|
+
raise UnsupportedDatatypeError(
|
|
891
|
+
f"{datatype!r} is a CLASS, not a data range — OWL 2 does not let "
|
|
892
|
+
f"one name be both, so it cannot be the datatype of a literal. "
|
|
893
|
+
f"The data domain's own 'everything' is rdfs:Literal; use a "
|
|
894
|
+
f"datatype name here")
|
|
895
|
+
return Literal(lexical, datatype)
|
|
896
|
+
|
|
897
|
+
|
|
898
|
+
|
|
899
|
+
# --------------------------------------------------------------------------- #
|
|
900
|
+
# Public API.
|
|
901
|
+
# --------------------------------------------------------------------------- #
|
|
902
|
+
|
|
903
|
+
def parse_manchester_literal(text: str) -> Literal:
|
|
904
|
+
"""Parse one Manchester-syntax literal: ``"400"^^xsd:integer``, ``"abc"``
|
|
905
|
+
(an ``xsd:string``), ``"abc"@en``, or a bare numeral (``400`` an
|
|
906
|
+
``xsd:integer``, ``1.5`` an ``xsd:decimal``, ``1.5f`` an ``xsd:float``).
|
|
907
|
+
|
|
908
|
+
Raises:
|
|
909
|
+
ManchesterSyntaxError: malformed text, an ill-typed literal.
|
|
910
|
+
"""
|
|
911
|
+
text = text.strip()
|
|
912
|
+
try:
|
|
913
|
+
if _bare_numeral(text):
|
|
914
|
+
return _bare_literal(text, None)
|
|
915
|
+
return _literal_from_token(text)
|
|
916
|
+
except UnsupportedDatatypeError as exc:
|
|
917
|
+
raise ManchesterSyntaxError(
|
|
918
|
+
f"parse_manchester_literal: {exc} in {text!r}") from exc
|
|
919
|
+
|
|
920
|
+
|
|
921
|
+
def parse_manchester_data_range(text: str, *, datatypes: Iterable[str] = ()) -> DataRange:
|
|
922
|
+
"""Parse ``text`` as a Manchester-syntax DATA RANGE: a datatype name,
|
|
923
|
+
``xsd:decimal[>= 10000, <= 30000]``, ``{1, 2}``, ``not DR``, ``DR and DR``,
|
|
924
|
+
``DR or DR`` and parentheses (precedence as for class expressions: ``not``
|
|
925
|
+
over ``and`` over ``or``).
|
|
926
|
+
|
|
927
|
+
Args:
|
|
928
|
+
text: The data range.
|
|
929
|
+
datatypes: Extra datatype names, for symmetry with
|
|
930
|
+
:func:`parse_manchester`; a name in a data-range position is a
|
|
931
|
+
datatype whether or not it is listed, so this only matters there.
|
|
932
|
+
|
|
933
|
+
Raises:
|
|
934
|
+
ManchesterSyntaxError: malformed text, or an out-of-scope facet
|
|
935
|
+
(``xsd:pattern``, the length facets, an ordering facet on a
|
|
936
|
+
non-numeric base) -- refused by name.
|
|
937
|
+
"""
|
|
938
|
+
return _Parser(_tokenize(text), text, datatypes).parse_data_range()
|
|
939
|
+
|
|
940
|
+
|
|
941
|
+
def to_manchester_data_range(datarange: DataRange) -> str:
|
|
942
|
+
"""Render ``datarange`` in Manchester Syntax, dual to
|
|
943
|
+
:func:`parse_manchester_data_range`: ``parse_manchester_data_range(
|
|
944
|
+
to_manchester_data_range(dr)) == dr`` for every data range. A facet bound is
|
|
945
|
+
written as a bare numeral exactly when reading that numeral back gives the
|
|
946
|
+
same literal; otherwise as a typed literal."""
|
|
947
|
+
return _render_datarange(datarange)
|
|
948
|
+
|
|
949
|
+
|
|
950
|
+
def parse_manchester(text: str, *, datatypes: Iterable[str] = ()) -> Concept:
|
|
951
|
+
"""Parse ``text`` (OWL 2 Manchester Syntax, ALC fragment) into a :class:`Concept`.
|
|
952
|
+
|
|
953
|
+
Round-trips against :func:`to_manchester`: ``parse_manchester(to_manchester(c))
|
|
954
|
+
== c`` for every ALC concept ``c`` (see the module docstring's "Round-trip
|
|
955
|
+
guarantee").
|
|
956
|
+
|
|
957
|
+
Args:
|
|
958
|
+
text: A Manchester-syntax class expression, e.g.
|
|
959
|
+
``"Person and hasChild some (Doctor and not Rich)"``.
|
|
960
|
+
datatypes: Extra datatype names to treat as data ranges, beyond the
|
|
961
|
+
OWL 2 datatype map's own. A restriction whose filler names one is a
|
|
962
|
+
DATA restriction (``HasNumber some OboRoIdrange1``); without the
|
|
963
|
+
declaration it reads as an object restriction over a class of that
|
|
964
|
+
name, because Manchester Syntax has no declaration table.
|
|
965
|
+
|
|
966
|
+
Returns:
|
|
967
|
+
The parsed :class:`Concept`.
|
|
968
|
+
|
|
969
|
+
Raises:
|
|
970
|
+
ManchesterSyntaxError: On malformed input (unbalanced parentheses, a
|
|
971
|
+
stray keyword, trailing garbage, …) or on syntax that is valid
|
|
972
|
+
Manchester/OWL 2 but outside ALC (cardinalities, ``value``,
|
|
973
|
+
``Self``, ``inverse``, nominals, datatype facets — see the module
|
|
974
|
+
docstring's "Rejected constructs").
|
|
975
|
+
"""
|
|
976
|
+
return _Parser(_tokenize(text), text, datatypes).parse_description()
|
|
977
|
+
|
|
978
|
+
|
|
979
|
+
def to_manchester(concept: Concept) -> str:
|
|
980
|
+
"""Render ``concept`` in OWL 2 Manchester Syntax, dual to :func:`parse_manchester`.
|
|
981
|
+
|
|
982
|
+
Parenthesises a child expression exactly when its precedence is below the
|
|
983
|
+
threshold of the slot it sits in (see the module docstring's "Grammar and
|
|
984
|
+
precedence"; the right operand of ``and`` / ``or`` is parenthesised when it
|
|
985
|
+
is itself an ``and`` / ``or``, since a flat chain reads to the left), so the
|
|
986
|
+
output is minimally parenthesised and re-parses to an identical AST.
|
|
987
|
+
|
|
988
|
+
Args:
|
|
989
|
+
concept: Any ALC :class:`Concept` (as built by
|
|
990
|
+
:mod:`unicode_logic_kit.dl.concepts`'s constructors).
|
|
991
|
+
|
|
992
|
+
Returns:
|
|
993
|
+
The Manchester-syntax rendering, e.g. ``"r some (A and B)"``.
|
|
994
|
+
|
|
995
|
+
Raises:
|
|
996
|
+
~unicode_logic_kit.dl.tableau.RoleExpressionError:
|
|
997
|
+
a restriction's role (at any depth) is an OWL 2
|
|
998
|
+
built-in property name or ``=`` / ``≠``. :func:`parse_manchester`
|
|
999
|
+
refuses that text, and
|
|
1000
|
+
:func:`~unicode_logic_kit.dl.owl_functional.to_owl_functional_class_expression`
|
|
1001
|
+
refuses the same concept, with the same function — a writer never
|
|
1002
|
+
prints text its own reader refuses.
|
|
1003
|
+
ValueError: a class, role, property or individual name has no spelling
|
|
1004
|
+
that :func:`parse_manchester` reads back as that name (see "Names
|
|
1005
|
+
with no spelling" in the module docstring), or an individual is
|
|
1006
|
+
spelled like a numeral.
|
|
1007
|
+
"""
|
|
1008
|
+
_reject_concept_roles_deep(concept, where="to_manchester")
|
|
1009
|
+
return _render(concept)
|
|
1010
|
+
|
|
1011
|
+
|
|
1012
|
+
def parse_manchester_axiom(text: str, *,
|
|
1013
|
+
datatypes: Iterable[str] = ()) -> Tuple[str, Concept, Concept]:
|
|
1014
|
+
"""Parse a Manchester-syntax subsumption or equivalence axiom.
|
|
1015
|
+
|
|
1016
|
+
Accepts exactly ``"C SubClassOf D"`` and ``"C EquivalentTo D"`` (the
|
|
1017
|
+
frame keyword may optionally carry its W3C-grammar trailing colon,
|
|
1018
|
+
``"SubClassOf:"``/``"EquivalentTo:"``), where ``C`` and ``D`` are each
|
|
1019
|
+
parsed by :func:`parse_manchester`. The keyword is located at
|
|
1020
|
+
parenthesis-depth 0; it must occur exactly once.
|
|
1021
|
+
|
|
1022
|
+
Args:
|
|
1023
|
+
text: An axiom of the form ``"<description> SubClassOf <description>"``
|
|
1024
|
+
or ``"<description> EquivalentTo <description>"``.
|
|
1025
|
+
|
|
1026
|
+
Returns:
|
|
1027
|
+
``("subclass", C, D)`` for ``C SubClassOf D``, or
|
|
1028
|
+
``("equivalent", C, D)`` for ``C EquivalentTo D``.
|
|
1029
|
+
|
|
1030
|
+
Raises:
|
|
1031
|
+
ManchesterSyntaxError: If no top-level ``SubClassOf``/``EquivalentTo``
|
|
1032
|
+
keyword is found, if more than one is found, or if either side
|
|
1033
|
+
fails to parse as an ALC description (see :func:`parse_manchester`).
|
|
1034
|
+
"""
|
|
1035
|
+
tokens = _tokenize(text)
|
|
1036
|
+
depth = 0
|
|
1037
|
+
found: List[Tuple[int, str]] = []
|
|
1038
|
+
for idx, (ttype, _value, _pos) in enumerate(tokens):
|
|
1039
|
+
if ttype == "LPAREN":
|
|
1040
|
+
depth += 1
|
|
1041
|
+
elif ttype == "RPAREN":
|
|
1042
|
+
depth -= 1
|
|
1043
|
+
elif depth == 0 and ttype in ("SUBCLASSOF", "EQUIVALENTTO"):
|
|
1044
|
+
found.append((idx, ttype))
|
|
1045
|
+
if not found:
|
|
1046
|
+
raise ManchesterSyntaxError(
|
|
1047
|
+
"parse_manchester_axiom: expected exactly one top-level "
|
|
1048
|
+
f"'SubClassOf' or 'EquivalentTo' keyword, found none in {text!r}")
|
|
1049
|
+
if len(found) > 1:
|
|
1050
|
+
raise ManchesterSyntaxError(
|
|
1051
|
+
"parse_manchester_axiom: expected exactly one top-level "
|
|
1052
|
+
f"'SubClassOf'/'EquivalentTo' keyword, found {len(found)} in {text!r}")
|
|
1053
|
+
idx, kind = found[0]
|
|
1054
|
+
left_tokens = tokens[:idx] + [("EOF", "", tokens[idx][2])]
|
|
1055
|
+
right_tokens = tokens[idx + 1:]
|
|
1056
|
+
sub = _Parser(left_tokens, text, datatypes).parse_description()
|
|
1057
|
+
sup = _Parser(right_tokens, text, datatypes).parse_description()
|
|
1058
|
+
label = "subclass" if kind == "SUBCLASSOF" else "equivalent"
|
|
1059
|
+
return (label, sub, sup)
|
|
1060
|
+
|
|
1061
|
+
|
|
1062
|
+
# Every OWL 2 object-property characteristic the W3C grammar recognises,
|
|
1063
|
+
# mapped to this module's own tag for it (the first element of
|
|
1064
|
+
# parse_manchester_role_axiom's return tuple). ALL SEVEN are read: a parser
|
|
1065
|
+
# never refuses an axiom KIND -- see "Role axioms" in the module docstring.
|
|
1066
|
+
_CHARACTERISTIC_TAGS = {
|
|
1067
|
+
"Transitive": "transitive",
|
|
1068
|
+
"Symmetric": "symmetric",
|
|
1069
|
+
"Asymmetric": "asymmetric",
|
|
1070
|
+
"Reflexive": "reflexive",
|
|
1071
|
+
"Irreflexive": "irreflexive",
|
|
1072
|
+
"Functional": "functional",
|
|
1073
|
+
"InverseFunctional": "inversefunctional",
|
|
1074
|
+
}
|
|
1075
|
+
|
|
1076
|
+
#: ``tag -> (frame keyword, arity)`` for the four BINARY role-axiom frames
|
|
1077
|
+
#: whose right-hand side is another ROLE NAME.
|
|
1078
|
+
_BINARY_ROLE_FRAMES = {
|
|
1079
|
+
"SUBPROPERTYOF": ("subproperty", "SubPropertyOf"),
|
|
1080
|
+
"EQUIVALENTTO": ("equivalentproperty", "EquivalentTo"),
|
|
1081
|
+
"INVERSEOF": ("inverse", "InverseOf"),
|
|
1082
|
+
"DISJOINTWITH": ("disjoint", "DisjointWith"),
|
|
1083
|
+
}
|
|
1084
|
+
|
|
1085
|
+
#: ``token -> (tag, frame keyword)`` for the two frames whose right-hand side
|
|
1086
|
+
#: is a CLASS EXPRESSION, not a role: ``r Domain: A`` and ``r Range: A``. A
|
|
1087
|
+
#: table of their own because the right side is parsed by ``_description()``
|
|
1088
|
+
#: and the returned tuple carries a :class:`Concept`, which is what widens
|
|
1089
|
+
#: :func:`parse_manchester_role_axiom`'s return type.
|
|
1090
|
+
_FILLER_ROLE_FRAMES = {
|
|
1091
|
+
"DOMAIN": ("domain", "Domain"),
|
|
1092
|
+
"RANGE": ("range", "Range"),
|
|
1093
|
+
}
|
|
1094
|
+
|
|
1095
|
+
_CHAIN_HINT = (
|
|
1096
|
+
"a PROPERTY CHAIN ('r o s SubPropertyOf t') is not one of this module's "
|
|
1097
|
+
"one-line role-axiom shapes: the bare name 'o' is deliberately not a "
|
|
1098
|
+
"keyword here, so a class or role literally named 'o' keeps working in "
|
|
1099
|
+
"parse_manchester. Read a chain with dl.parse_owl_functional "
|
|
1100
|
+
"('SubObjectPropertyOf(ObjectPropertyChain(r s) t)') or build it with "
|
|
1101
|
+
"dl.TBox.add_role_chain")
|
|
1102
|
+
|
|
1103
|
+
|
|
1104
|
+
def _check_role_axiom_name(role: str, where: str) -> None:
|
|
1105
|
+
"""Refuse an OWL 2 BUILT-IN property name in a Manchester role axiom.
|
|
1106
|
+
|
|
1107
|
+
Refused in EVERY position, including the tautological super-role case
|
|
1108
|
+
``dl.parse_owl_functional`` consumes as a no-op: a single-axiom parser has
|
|
1109
|
+
no return shape for "this axiom is nothing". Deliberate asymmetry between
|
|
1110
|
+
the two parsers, documented in both (see "Role axioms" in this module's
|
|
1111
|
+
docstring and "The OWL 2 built-in roles" in ``dl.tableau``'s).
|
|
1112
|
+
"""
|
|
1113
|
+
builtin = reserved_role(role)
|
|
1114
|
+
if builtin is None:
|
|
1115
|
+
return
|
|
1116
|
+
universal = builtin in RESERVED_TOP_ROLES
|
|
1117
|
+
raise ManchesterSyntaxError(
|
|
1118
|
+
f"parse_manchester_role_axiom: {role!r} is an OWL 2 BUILT-IN property "
|
|
1119
|
+
f"({builtin} — "
|
|
1120
|
+
+ ("the universal property: it relates every pair"
|
|
1121
|
+
if universal else "the empty property: it relates no pair at all")
|
|
1122
|
+
+ f"), not an ordinary role name, and ALCHQ (this kit's DL fragment) "
|
|
1123
|
+
f"has neither the universal nor the empty role, so {where} cannot "
|
|
1124
|
+
f"carry it. "
|
|
1125
|
+
+ ("An inclusion INTO it is a TAUTOLOGY, which this single-axiom "
|
|
1126
|
+
"parser has no return shape for — dl.parse_owl_functional reads "
|
|
1127
|
+
"that shape and consumes it as a documented no-op."
|
|
1128
|
+
if universal else
|
|
1129
|
+
"'P is empty' is the concept inclusion 'owl:Thing SubClassOf P only "
|
|
1130
|
+
"owl:Nothing', which parse_manchester_axiom does read.")
|
|
1131
|
+
)
|
|
1132
|
+
|
|
1133
|
+
|
|
1134
|
+
def parse_manchester_role_axiom(text: str) -> Tuple[object, ...]:
|
|
1135
|
+
"""Parse one Manchester-syntax role-box axiom.
|
|
1136
|
+
|
|
1137
|
+
Seven one-line shapes, of the full W3C ``ObjectProperty:`` frame syntax —
|
|
1138
|
+
see "Role axioms" in the module docstring for why only these, and for the
|
|
1139
|
+
two shapes that stay refused by name. Every frame keyword may carry its
|
|
1140
|
+
W3C-grammar trailing colon (``"SubPropertyOf:"``), exactly like
|
|
1141
|
+
``SubClassOf``/``SubClassOf:`` in :func:`parse_manchester_axiom`.
|
|
1142
|
+
|
|
1143
|
+
Args:
|
|
1144
|
+
text: A single role axiom, e.g. ``"hasChild SubPropertyOf hasDescendant"``,
|
|
1145
|
+
``"partOf InverseOf hasPart"``, ``"hasSink DisjointWith hasSource"``,
|
|
1146
|
+
``"hasSink EquivalentTo hasOutput"``,
|
|
1147
|
+
``"hasDescendant Characteristics: Transitive"``,
|
|
1148
|
+
``"Covers Domain: Study"`` or ``"HasUnit Range: Unit and Measurable"``.
|
|
1149
|
+
|
|
1150
|
+
Returns:
|
|
1151
|
+
``("subproperty", sub_role, super_role)`` (feeding
|
|
1152
|
+
:meth:`~unicode_logic_kit.dl.tableau.TBox.add_role_inclusion`),
|
|
1153
|
+
``("equivalentproperty", p, q)`` (``add_equivalent_roles``),
|
|
1154
|
+
``("inverse", p, q)`` (``add_inverse_roles``),
|
|
1155
|
+
``("disjoint", p, q)`` (``add_disjoint_roles``),
|
|
1156
|
+
``("domain", role, Concept)`` (``add_role_domain``),
|
|
1157
|
+
``("range", role, Concept)`` (``add_role_range``), or
|
|
1158
|
+
``(tag, role)`` for a ``Characteristics:`` declaration, where ``tag``
|
|
1159
|
+
is one of :data:`_CHARACTERISTIC_TAGS`' values and
|
|
1160
|
+
``"add_" + tag + "_role"`` is the builder — except
|
|
1161
|
+
``"inversefunctional"``, whose builder is
|
|
1162
|
+
``add_inverse_functional_role``.
|
|
1163
|
+
|
|
1164
|
+
The return type is ``Tuple[object, ...]``, not ``Tuple[str, ...]``: the
|
|
1165
|
+
``Domain:``/``Range:`` shapes carry a :class:`Concept` in the third slot,
|
|
1166
|
+
because a domain or range axiom's right-hand side IS a class expression —
|
|
1167
|
+
parsed by the same ``_description()`` every other class-expression position
|
|
1168
|
+
uses, so ``"r Domain: A and B"`` works.
|
|
1169
|
+
|
|
1170
|
+
Raises:
|
|
1171
|
+
ManchesterSyntaxError: malformed input, an unknown role characteristic
|
|
1172
|
+
(named explicitly in the message), an OWL 2 built-in property
|
|
1173
|
+
name, a property chain, or anything not matching one of the seven
|
|
1174
|
+
shapes.
|
|
1175
|
+
"""
|
|
1176
|
+
tokens = _tokenize(text)
|
|
1177
|
+
|
|
1178
|
+
def error(message: str) -> ManchesterSyntaxError:
|
|
1179
|
+
return ManchesterSyntaxError(f"parse_manchester_role_axiom: {message} in {text!r}")
|
|
1180
|
+
|
|
1181
|
+
if tokens[0][0] != "NAME":
|
|
1182
|
+
found = "end of input" if tokens[0][0] == "EOF" else f"{tokens[0][1]!r}"
|
|
1183
|
+
raise error(f"expected a role name but found {found} at position {tokens[0][2]}")
|
|
1184
|
+
role = tokens[0][1]
|
|
1185
|
+
keyword = tokens[1]
|
|
1186
|
+
|
|
1187
|
+
if keyword[0] in _BINARY_ROLE_FRAMES:
|
|
1188
|
+
tag, spelling = _BINARY_ROLE_FRAMES[keyword[0]]
|
|
1189
|
+
if len(tokens) == 4 and tokens[2][0] == "NAME" and tokens[3][0] == "EOF":
|
|
1190
|
+
_check_role_axiom_name(role, f"a {spelling} axiom")
|
|
1191
|
+
_check_role_axiom_name(tokens[2][1], f"a {spelling} axiom")
|
|
1192
|
+
return (tag, role, tokens[2][1])
|
|
1193
|
+
if (keyword[0] in ("DISJOINTWITH", "EQUIVALENTTO")
|
|
1194
|
+
and any(t[0] == "NAME" for t in tokens[2:])
|
|
1195
|
+
and text.count(",") > 0):
|
|
1196
|
+
raise error(
|
|
1197
|
+
f"the comma-separated n-ary {spelling} frame slot is not one "
|
|
1198
|
+
f"of this module's one-line role-axiom shapes — only the "
|
|
1199
|
+
f"binary '<role> {spelling} <role>' is. Read the n-ary form "
|
|
1200
|
+
f"with dl.parse_owl_functional, which expands it correctly "
|
|
1201
|
+
f"(DisjointObjectProperties needs ALL pairs, not a chain of "
|
|
1202
|
+
f"consecutive ones), or call dl.TBox."
|
|
1203
|
+
+ ("add_disjoint_roles" if keyword[0] == "DISJOINTWITH"
|
|
1204
|
+
else "add_equivalent_roles")
|
|
1205
|
+
+ " with every role")
|
|
1206
|
+
raise error(f"expected exactly '<role> {spelling} <role>'")
|
|
1207
|
+
|
|
1208
|
+
if keyword[0] in _FILLER_ROLE_FRAMES:
|
|
1209
|
+
tag, spelling = _FILLER_ROLE_FRAMES[keyword[0]]
|
|
1210
|
+
_check_role_axiom_name(role, f"a {spelling}: axiom")
|
|
1211
|
+
if tokens[2][0] == "EOF":
|
|
1212
|
+
raise error(f"expected exactly '<role> {spelling}: <description>'")
|
|
1213
|
+
# The filler goes through the full description grammar, so
|
|
1214
|
+
# `r Domain: A and B` and `r Range: s some C` both read -- a domain or
|
|
1215
|
+
# range axiom's right-hand side is a CLASS EXPRESSION, not a name.
|
|
1216
|
+
filler = _Parser(list(tokens[2:]), text).parse_description()
|
|
1217
|
+
return (tag, role, filler)
|
|
1218
|
+
|
|
1219
|
+
if keyword[0] == "CHARACTERISTICS":
|
|
1220
|
+
if len(tokens) == 4 and tokens[2][0] == "NAME" and tokens[3][0] == "EOF":
|
|
1221
|
+
characteristic = tokens[2][1]
|
|
1222
|
+
if characteristic in _CHARACTERISTIC_TAGS:
|
|
1223
|
+
_check_role_axiom_name(role, f"a Characteristics: "
|
|
1224
|
+
f"{characteristic} axiom")
|
|
1225
|
+
return (_CHARACTERISTIC_TAGS[characteristic], role)
|
|
1226
|
+
raise error(
|
|
1227
|
+
f"unknown role characteristic {characteristic!r} — OWL 2 has "
|
|
1228
|
+
f"exactly seven: " + ", ".join(sorted(_CHARACTERISTIC_TAGS)))
|
|
1229
|
+
raise error("expected exactly '<role> Characteristics: <characteristic>'")
|
|
1230
|
+
|
|
1231
|
+
if keyword[0] == "NAME" and keyword[1] == "o":
|
|
1232
|
+
raise error(_CHAIN_HINT)
|
|
1233
|
+
|
|
1234
|
+
found = "end of input" if keyword[0] == "EOF" else f"{keyword[1]!r}"
|
|
1235
|
+
raise error(
|
|
1236
|
+
f"expected 'SubPropertyOf', 'EquivalentTo', 'InverseOf', "
|
|
1237
|
+
f"'DisjointWith', 'Domain:', 'Range:' or 'Characteristics:' but found "
|
|
1238
|
+
f"{found} at position {keyword[2]}")
|
|
1239
|
+
|
|
1240
|
+
|
|
1241
|
+
# --------------------------------------------------------------------------- #
|
|
1242
|
+
# Renderer.
|
|
1243
|
+
# --------------------------------------------------------------------------- #
|
|
1244
|
+
|
|
1245
|
+
# Same lattice as concepts.py's _PREC (Or=1 < And=2 < Not=Exists=ForAll=AtLeast=
|
|
1246
|
+
# AtMost=3 < Atomic=4): see the module docstring's "Grammar and precedence" for why
|
|
1247
|
+
# the two coincide.
|
|
1248
|
+
_PREC = {Or: 1, And: 2, Not: 3, Exists: 3, ForAll: 3, AtLeast: 3, AtMost: 3,
|
|
1249
|
+
HasValue: 3, DataExists: 3, DataForAll: 3, DataHasValue: 3,
|
|
1250
|
+
DataAtLeast: 3, DataAtMost: 3,
|
|
1251
|
+
Atomic: 4, Top: 4, Bottom: 4, Nominal: 4}
|
|
1252
|
+
|
|
1253
|
+
|
|
1254
|
+
def _one_name_token(spelled: str, name: str) -> Optional[List[_Token]]:
|
|
1255
|
+
"""The tokens of ``spelled`` when the reader takes it for ONE plain name that
|
|
1256
|
+
is exactly ``name`` (not a keyword, not a literal, not several words), else
|
|
1257
|
+
``None``."""
|
|
1258
|
+
try:
|
|
1259
|
+
tokens = _tokenize(spelled)
|
|
1260
|
+
except ManchesterSyntaxError:
|
|
1261
|
+
return None
|
|
1262
|
+
if len(tokens) == 2 and tokens[0][0] == "NAME" and tokens[0][1] == name:
|
|
1263
|
+
return tokens
|
|
1264
|
+
return None
|
|
1265
|
+
|
|
1266
|
+
|
|
1267
|
+
def _why_no_spelling(name: str) -> str:
|
|
1268
|
+
"""Why no spelling of ``name`` is read back as that one name."""
|
|
1269
|
+
if name == "":
|
|
1270
|
+
return "it is empty"
|
|
1271
|
+
if _FULL_IRI.fullmatch(name):
|
|
1272
|
+
return ("it already holds the angle brackets of a full IRI, and the reader "
|
|
1273
|
+
"stores an IRI without them, so <...> reads back as another name "
|
|
1274
|
+
"(the text between the brackets)")
|
|
1275
|
+
if any(ch.isspace() for ch in name):
|
|
1276
|
+
return ("it holds whitespace, where the reader splits a name, and only a "
|
|
1277
|
+
"full IRI (a scheme, no whitespace) is written in angle brackets")
|
|
1278
|
+
if _classify_word(name) != "NAME":
|
|
1279
|
+
return ("it is a keyword of this syntax, which has no escape for it: only "
|
|
1280
|
+
"a full IRI (a scheme, no whitespace) is written in angle brackets")
|
|
1281
|
+
if any(ch in _STRUCT_TOKENS for ch in name):
|
|
1282
|
+
return ("it holds one of ( ) { } [ ] , and has no scheme to be bracketed "
|
|
1283
|
+
"with as a full IRI")
|
|
1284
|
+
return "no spelling of it is read back as one name"
|
|
1285
|
+
|
|
1286
|
+
|
|
1287
|
+
def _names_a_builtin_datatype(name: str) -> bool:
|
|
1288
|
+
"""True iff ``name`` is a built-in datatype of the OWL 2 datatype map, in
|
|
1289
|
+
either spelling of its namespace (``xsd:integer`` or the full IRI, bracketed
|
|
1290
|
+
or not) — what the reader takes for a datatype wherever a datatype can stand."""
|
|
1291
|
+
return canonical_datatype_name(name) in BUILTIN_DATATYPES
|
|
1292
|
+
|
|
1293
|
+
|
|
1294
|
+
_DATATYPE_CLASS_REASON = (
|
|
1295
|
+
"it is the name of a built-in datatype, and the reader reads that name as the "
|
|
1296
|
+
"datatype in every spelling (the full IRI and either namespace form included): "
|
|
1297
|
+
"after some / only / min / max / exactly it turns the restriction into a DATA "
|
|
1298
|
+
"restriction, and in any other position it refuses the text, because OWL 2 "
|
|
1299
|
+
"does not let one name be both a class and a datatype")
|
|
1300
|
+
|
|
1301
|
+
|
|
1302
|
+
@functools.lru_cache(maxsize=8192)
|
|
1303
|
+
def _name_spelling(name: str, kind: str) -> Tuple[Optional[str], str]:
|
|
1304
|
+
"""``(spelling, "")`` for the one-token spelling of ``name`` that the reader
|
|
1305
|
+
reads back as exactly that name, or ``(None, reason)`` when there is none.
|
|
1306
|
+
|
|
1307
|
+
The spellings tried are the full IRI ``<name>`` (first, for a name that holds
|
|
1308
|
+
``://`` or a structural character, as before) and the bare name; a spelling
|
|
1309
|
+
counts only when the READER's own tokenizer takes it for one plain name, and,
|
|
1310
|
+
for a ``kind`` of ``"class"`` / ``"individual"``, when the reader's own
|
|
1311
|
+
parser then reads it back as that class / that individual and not as another
|
|
1312
|
+
expression (``owl:Thing`` is the top class in every spelling; a numeral is a
|
|
1313
|
+
data value).
|
|
1314
|
+
|
|
1315
|
+
A class named like a built-in datatype has no spelling at all. The reader
|
|
1316
|
+
decides between an object and a data restriction by the FILLER, so
|
|
1317
|
+
``r some xsd:integer`` is a data restriction whatever the writer meant, and
|
|
1318
|
+
the full IRI reads the same way: refusing the bare class alone, which the
|
|
1319
|
+
reader refuses on its own, would leave the same name written, and read as
|
|
1320
|
+
something else, in the one position where it is a filler.
|
|
1321
|
+
"""
|
|
1322
|
+
candidates: List[str] = []
|
|
1323
|
+
bracketed = f"<{name}>"
|
|
1324
|
+
if ("://" in name or any(ch in _STRUCT_TOKENS for ch in name)) \
|
|
1325
|
+
and _FULL_IRI.fullmatch(bracketed):
|
|
1326
|
+
candidates.append(bracketed)
|
|
1327
|
+
candidates.append(name)
|
|
1328
|
+
for spelled in candidates:
|
|
1329
|
+
tokens = _one_name_token(spelled, name)
|
|
1330
|
+
if tokens is None:
|
|
1331
|
+
continue
|
|
1332
|
+
if kind == "class":
|
|
1333
|
+
if _names_a_builtin_datatype(name):
|
|
1334
|
+
return None, _DATATYPE_CLASS_REASON
|
|
1335
|
+
try:
|
|
1336
|
+
back = _Parser(tokens, spelled).parse_description()
|
|
1337
|
+
except ManchesterSyntaxError:
|
|
1338
|
+
return spelled, "" # refused on reading: loud, not another reading
|
|
1339
|
+
if back != Atomic(name):
|
|
1340
|
+
return None, (f"read as a class it is {back!r}, in every spelling "
|
|
1341
|
+
f"(owl:Thing and owl:Nothing are the top and the bottom class)")
|
|
1342
|
+
elif kind == "individual":
|
|
1343
|
+
text = f"r value {spelled}"
|
|
1344
|
+
try:
|
|
1345
|
+
back = _Parser(_tokenize(text), text).parse_description()
|
|
1346
|
+
except ManchesterSyntaxError:
|
|
1347
|
+
return spelled, ""
|
|
1348
|
+
if back != HasValue("r", name):
|
|
1349
|
+
return None, f"read after 'value' it is {back!r}, not an individual"
|
|
1350
|
+
return spelled, ""
|
|
1351
|
+
return None, _why_no_spelling(name)
|
|
1352
|
+
|
|
1353
|
+
|
|
1354
|
+
def _render_name(name: str, kind: str = "name") -> str:
|
|
1355
|
+
"""A kit name as ONE Manchester token that reads back as that name: ``<name>``
|
|
1356
|
+
when it is a full IRI (it holds ``://``) or holds a structural character a
|
|
1357
|
+
bare name could not carry, the bare name otherwise. The brackets are added
|
|
1358
|
+
here and only here (the readers store an IRI without them).
|
|
1359
|
+
|
|
1360
|
+
``kind`` is ``"class"`` or ``"individual"`` where the reader gives the name a
|
|
1361
|
+
meaning of its own (``owl:Thing``, a numeral), and ``"name"`` elsewhere.
|
|
1362
|
+
|
|
1363
|
+
Raises:
|
|
1364
|
+
ValueError: no spelling reads back as ``name``: it is a keyword of the
|
|
1365
|
+
syntax, holds whitespace or a structural character that no full IRI
|
|
1366
|
+
can carry (an IRI needs a scheme and no whitespace), is empty, already
|
|
1367
|
+
holds the brackets of a full IRI (which the reader strips), is
|
|
1368
|
+
read as another expression (``owl:Thing``), or is a built-in
|
|
1369
|
+
datatype's name used as a class (the reader makes a restriction
|
|
1370
|
+
over it a data restriction). The text is never written as
|
|
1371
|
+
something that reads back as another name or expression.
|
|
1372
|
+
"""
|
|
1373
|
+
spelling, reason = _name_spelling(name, kind)
|
|
1374
|
+
if spelling is None:
|
|
1375
|
+
if reason == _DATATYPE_CLASS_REASON:
|
|
1376
|
+
remedy = ("Rename the class: every reader of this kit reads a "
|
|
1377
|
+
"built-in datatype's name as that datatype, or refuses it as "
|
|
1378
|
+
"a class.")
|
|
1379
|
+
else:
|
|
1380
|
+
remedy = ("Rename it, or write the concept with "
|
|
1381
|
+
"dl.to_owl_functional_class_expression, whose <...> carries a "
|
|
1382
|
+
"name that has no '>' in it.")
|
|
1383
|
+
raise ValueError(
|
|
1384
|
+
f"to_manchester: the name {name!r} cannot be written so that "
|
|
1385
|
+
f"parse_manchester reads it back as that name: {reason}. {remedy}")
|
|
1386
|
+
return spelling
|
|
1387
|
+
|
|
1388
|
+
|
|
1389
|
+
def _render_literal(literal: Literal) -> str:
|
|
1390
|
+
"""A literal as one Manchester token: ``"5"^^xsd:integer`` with the datatype
|
|
1391
|
+
written through :func:`_render_name`, so a user datatype IRI is bracketed
|
|
1392
|
+
(``"5"^^<http://ex.org/dt,Small>``) and reads back as the same datatype."""
|
|
1393
|
+
return render_literal_fs(literal, _render_name)
|
|
1394
|
+
|
|
1395
|
+
|
|
1396
|
+
def _render_role(role) -> str:
|
|
1397
|
+
"""Render a ``role`` field: ``inverse r`` for an
|
|
1398
|
+
:class:`~unicode_logic_kit.dl.concepts.InverseRole` (the W3C grammar's own
|
|
1399
|
+
spelling — see "Export-only asymmetry" in the module docstring), the bare
|
|
1400
|
+
name otherwise.
|
|
1401
|
+
"""
|
|
1402
|
+
if isinstance(role, InverseRole):
|
|
1403
|
+
return f"inverse {_render_name(role.role)}"
|
|
1404
|
+
return _render_name(role)
|
|
1405
|
+
|
|
1406
|
+
|
|
1407
|
+
def _render(c: Concept) -> str:
|
|
1408
|
+
"""Render a concept with precedence-aware parenthesisation."""
|
|
1409
|
+
if isinstance(c, Top):
|
|
1410
|
+
return "owl:Thing"
|
|
1411
|
+
if isinstance(c, Bottom):
|
|
1412
|
+
return "owl:Nothing"
|
|
1413
|
+
if isinstance(c, Atomic):
|
|
1414
|
+
return _render_name(c.name, "class")
|
|
1415
|
+
if isinstance(c, Nominal):
|
|
1416
|
+
# Behind `some` / `only` / `min` / `max` the reader takes `{3}` for a set of
|
|
1417
|
+
# DATA values, not for a nominal, so an individual spelled like a numeral
|
|
1418
|
+
# has no nominal spelling that reads back as it (cf. the value restriction).
|
|
1419
|
+
if _bare_numeral(c.individual):
|
|
1420
|
+
raise ValueError(
|
|
1421
|
+
f"to_manchester: the individual {c.individual!r} of the nominal "
|
|
1422
|
+
f"{c.to_unicode()} is spelled like a numeral, and Manchester Syntax "
|
|
1423
|
+
f"reads '{{{c.individual}}}' behind a role keyword as a set of DATA "
|
|
1424
|
+
f"values. There is no spelling that reads back as this individual: "
|
|
1425
|
+
f"rename it, or write the concept with "
|
|
1426
|
+
f"dl.to_owl_functional_class_expression.")
|
|
1427
|
+
return "{" + _render_name(c.individual) + "}"
|
|
1428
|
+
if isinstance(c, Not):
|
|
1429
|
+
return "not " + _paren(c.concept, 3)
|
|
1430
|
+
# The reader folds a flat chain LEFT (``A and B and C`` is ``(A and B) and C``),
|
|
1431
|
+
# so only a LEFT operand of the same connective may go unparenthesised: the
|
|
1432
|
+
# right operand is written one level tighter, and an ``and`` inside an
|
|
1433
|
+
# ``and`` (an ``or`` inside an ``or``) on the right keeps its parentheses.
|
|
1434
|
+
if isinstance(c, And):
|
|
1435
|
+
return f"{_paren(c.left, 2)} and {_paren(c.right, 3)}"
|
|
1436
|
+
if isinstance(c, Or):
|
|
1437
|
+
return f"{_paren(c.left, 1)} or {_paren(c.right, 2)}"
|
|
1438
|
+
if isinstance(c, Exists):
|
|
1439
|
+
return f"{_render_role(c.role)} some {_paren(c.concept, 3)}"
|
|
1440
|
+
if isinstance(c, ForAll):
|
|
1441
|
+
return f"{_render_role(c.role)} only {_paren(c.concept, 3)}"
|
|
1442
|
+
if isinstance(c, HasValue):
|
|
1443
|
+
# `r value 5` and `r value 1.5f` are DATA value restrictions to the
|
|
1444
|
+
# reader: an individual whose name is spelled like a bare numeral cannot
|
|
1445
|
+
# be written so that it reads back as an individual. Refused by name,
|
|
1446
|
+
# like every other text this writer would not read back as it was.
|
|
1447
|
+
if _bare_numeral(c.individual):
|
|
1448
|
+
raise ValueError(
|
|
1449
|
+
f"to_manchester: the individual {c.individual!r} of the value "
|
|
1450
|
+
f"restriction {c.to_unicode()} is spelled like a numeral, and "
|
|
1451
|
+
f"Manchester Syntax reads '{_render_role(c.role)} value "
|
|
1452
|
+
f"{c.individual}' as a DATA value restriction to that literal. "
|
|
1453
|
+
f"There is no spelling that reads back as this individual: "
|
|
1454
|
+
f"rename it, or write the axiom with dl.to_owl_functional.")
|
|
1455
|
+
# No _paren: the operand is an individual NAME, never a nested
|
|
1456
|
+
# description, so there is no precedence question to ask.
|
|
1457
|
+
return f"{_render_role(c.role)} value {_render_name(c.individual, 'individual')}"
|
|
1458
|
+
if isinstance(c, AtLeast):
|
|
1459
|
+
return _number_restriction_render(c.role, "min", c.n, c.concept)
|
|
1460
|
+
if isinstance(c, AtMost):
|
|
1461
|
+
return _number_restriction_render(c.role, "max", c.n, c.concept)
|
|
1462
|
+
# The data restrictions. The filler of a data cardinality is ALWAYS written
|
|
1463
|
+
# (even rdfs:Literal): `d min 2` alone reads back as an OBJECT restriction,
|
|
1464
|
+
# because with no filler there is nothing to say it is a data property.
|
|
1465
|
+
if isinstance(c, DataExists):
|
|
1466
|
+
return f"{_render_name(c.prop)} some {_render_datarange(c.datarange, paren=True)}"
|
|
1467
|
+
if isinstance(c, DataForAll):
|
|
1468
|
+
return f"{_render_name(c.prop)} only {_render_datarange(c.datarange, paren=True)}"
|
|
1469
|
+
if isinstance(c, DataHasValue):
|
|
1470
|
+
return f"{_render_name(c.prop)} value {_render_literal(c.value)}"
|
|
1471
|
+
if isinstance(c, DataAtLeast):
|
|
1472
|
+
return f"{_render_name(c.prop)} min {c.n} {_render_datarange(c.datarange, paren=True)}"
|
|
1473
|
+
if isinstance(c, DataAtMost):
|
|
1474
|
+
return f"{_render_name(c.prop)} max {c.n} {_render_datarange(c.datarange, paren=True)}"
|
|
1475
|
+
raise TypeError(f"to_manchester: unsupported concept {type(c).__name__}")
|
|
1476
|
+
|
|
1477
|
+
|
|
1478
|
+
def _render_bound(bound: Literal, base: str) -> str:
|
|
1479
|
+
"""A facet bound as a bare numeral if reading it back gives the same
|
|
1480
|
+
literal, else as a typed literal."""
|
|
1481
|
+
bare = bound.lexical.strip()
|
|
1482
|
+
if _bare_numeral(bare):
|
|
1483
|
+
try:
|
|
1484
|
+
if _bare_literal(bare, base) == bound:
|
|
1485
|
+
return bare
|
|
1486
|
+
except UnsupportedDatatypeError:
|
|
1487
|
+
pass
|
|
1488
|
+
return _render_literal(bound)
|
|
1489
|
+
|
|
1490
|
+
|
|
1491
|
+
def _render_datarange(dr: DataRange, *, paren: bool = False) -> str:
|
|
1492
|
+
"""Render a data range; ``paren`` parenthesises a compound one, for a slot
|
|
1493
|
+
(a restriction's filler, an ``and``/``or`` operand) that binds tighter."""
|
|
1494
|
+
if isinstance(dr, Datatype):
|
|
1495
|
+
return _render_name(dr.name)
|
|
1496
|
+
if isinstance(dr, DatatypeRestriction):
|
|
1497
|
+
facets = ", ".join(f"{_SYMBOL_OF_FACET[facet]} {_render_bound(bound, dr.base.name)}"
|
|
1498
|
+
for facet, bound in dr.facets)
|
|
1499
|
+
return f"{_render_name(dr.base.name)}[{facets}]"
|
|
1500
|
+
if isinstance(dr, DataOneOf):
|
|
1501
|
+
return "{" + ", ".join(_render_bound(v, "") for v in dr.values) + "}"
|
|
1502
|
+
if isinstance(dr, DataComplementOf):
|
|
1503
|
+
return "not " + _render_datarange(dr.datarange, paren=True)
|
|
1504
|
+
if isinstance(dr, DataIntersectionOf):
|
|
1505
|
+
inner = " and ".join(_render_datarange(r, paren=True) for r in dr.ranges)
|
|
1506
|
+
return f"({inner})" if paren else inner
|
|
1507
|
+
if isinstance(dr, DataUnionOf):
|
|
1508
|
+
inner = " or ".join(_render_datarange(r, paren=True) for r in dr.ranges)
|
|
1509
|
+
return f"({inner})" if paren else inner
|
|
1510
|
+
raise TypeError(f"to_manchester: unsupported data range {type(dr).__name__}")
|
|
1511
|
+
|
|
1512
|
+
|
|
1513
|
+
def _number_restriction_render(role, keyword: str, n: int, filler: Concept) -> str:
|
|
1514
|
+
"""Render ``role (min|max) n filler``, omitting the qualifying class entirely
|
|
1515
|
+
when ``filler`` is ⊤ (the "unqualified restriction" shorthand — see the module
|
|
1516
|
+
docstring's "Qualified number restrictions"; both spellings parse back to the
|
|
1517
|
+
same ``Top()``-qualified concept).
|
|
1518
|
+
"""
|
|
1519
|
+
rendered_role = _render_role(role)
|
|
1520
|
+
if isinstance(filler, Top):
|
|
1521
|
+
return f"{rendered_role} {keyword} {n}"
|
|
1522
|
+
return f"{rendered_role} {keyword} {n} {_paren(filler, 3)}"
|
|
1523
|
+
|
|
1524
|
+
|
|
1525
|
+
def _paren(c: Concept, parent_prec: int) -> str:
|
|
1526
|
+
"""Parenthesise ``c`` when its precedence is below the parent slot's threshold."""
|
|
1527
|
+
inner = _render(c)
|
|
1528
|
+
return f"({inner})" if _PREC.get(type(c), 4) < parent_prec else inner
|
|
1529
|
+
|
|
1530
|
+
|
|
1531
|
+
#: ``tag -> frame keyword`` for the four binary shapes, derived from
|
|
1532
|
+
#: :data:`_BINARY_ROLE_FRAMES` so the reader and the writer cannot name the
|
|
1533
|
+
#: same shape differently.
|
|
1534
|
+
_BINARY_FRAME_SPELLING = {tag: spelling
|
|
1535
|
+
for tag, spelling in _BINARY_ROLE_FRAMES.values()}
|
|
1536
|
+
|
|
1537
|
+
#: ``tag -> frame keyword`` for the two CLASS-EXPRESSION frames, derived from
|
|
1538
|
+
#: :data:`_FILLER_ROLE_FRAMES` for the same reason.
|
|
1539
|
+
_FILLER_FRAME_SPELLING = {tag: spelling
|
|
1540
|
+
for tag, spelling in _FILLER_ROLE_FRAMES.values()}
|
|
1541
|
+
|
|
1542
|
+
#: ``tag -> characteristic word``, the inverse of :data:`_CHARACTERISTIC_TAGS`.
|
|
1543
|
+
_TAG_CHARACTERISTIC = {tag: word for word, tag in _CHARACTERISTIC_TAGS.items()}
|
|
1544
|
+
|
|
1545
|
+
|
|
1546
|
+
def _renderable_role(role: object, axiom: Tuple[object, ...]) -> str:
|
|
1547
|
+
"""The role operand of a role axiom, or a ``ValueError`` naming why not.
|
|
1548
|
+
|
|
1549
|
+
The reader takes a plain role NAME in every position (an inverse role has
|
|
1550
|
+
no one-line spelling here, and the OWL 2 built-ins are refused by name), so
|
|
1551
|
+
writing anything else would print text that does not read back -- for an
|
|
1552
|
+
``InverseRole`` the dataclass ``repr``, ``InverseRole(role='s')``, which
|
|
1553
|
+
no reader would take for a role.
|
|
1554
|
+
"""
|
|
1555
|
+
if not isinstance(role, str):
|
|
1556
|
+
raise ValueError(
|
|
1557
|
+
f"role_axiom_to_manchester: a role operand must be a role NAME (a "
|
|
1558
|
+
f"str), got {role!r} in {axiom!r}. The one-line role-axiom syntax "
|
|
1559
|
+
f"has no spelling for an inverse role (dl.parse_manchester_role_axiom "
|
|
1560
|
+
f"reads none): write it with dl.to_owl_functional, which spells a "
|
|
1561
|
+
f"role inclusion with an inverse as SubObjectPropertyOf(r "
|
|
1562
|
+
f"ObjectInverseOf(s)).")
|
|
1563
|
+
builtin = reserved_role(role)
|
|
1564
|
+
if builtin is not None:
|
|
1565
|
+
raise ValueError(
|
|
1566
|
+
f"role_axiom_to_manchester: {role!r} is an OWL 2 BUILT-IN property "
|
|
1567
|
+
f"({builtin}), not an ordinary role name -- "
|
|
1568
|
+
f"dl.parse_manchester_role_axiom refuses it in every position, so "
|
|
1569
|
+
f"the text written here would not read back, in {axiom!r}.")
|
|
1570
|
+
return _render_name(role)
|
|
1571
|
+
|
|
1572
|
+
|
|
1573
|
+
def role_axiom_to_manchester(*axiom: object) -> str:
|
|
1574
|
+
"""Render a role-box axiom, dual to :func:`parse_manchester_role_axiom`.
|
|
1575
|
+
|
|
1576
|
+
Round-trips: ``parse_manchester_role_axiom(role_axiom_to_manchester(*axiom))
|
|
1577
|
+
== axiom`` for every ``axiom`` one of that function's return shapes. All
|
|
1578
|
+
three tables are derived from the reader's own (see
|
|
1579
|
+
:data:`_BINARY_FRAME_SPELLING`), so a shape the reader gains cannot be
|
|
1580
|
+
spelled differently here.
|
|
1581
|
+
|
|
1582
|
+
Args:
|
|
1583
|
+
*axiom: One of ``("subproperty", sub, sup)``,
|
|
1584
|
+
``("equivalentproperty", p, q)``, ``("inverse", p, q)``,
|
|
1585
|
+
``("disjoint", p, q)``, ``("domain", role, Concept)``,
|
|
1586
|
+
``("range", role, Concept)`` or ``(characteristic_tag, role)`` —
|
|
1587
|
+
exactly what :func:`parse_manchester_role_axiom` returns.
|
|
1588
|
+
|
|
1589
|
+
Returns:
|
|
1590
|
+
The one-line Manchester spelling, e.g.
|
|
1591
|
+
``"partOf InverseOf hasPart"``, ``"r Characteristics: Asymmetric"`` or
|
|
1592
|
+
``"Covers Domain: Study"``.
|
|
1593
|
+
|
|
1594
|
+
Raises:
|
|
1595
|
+
ValueError: ``axiom`` is not one of the recognised shapes, or a role
|
|
1596
|
+
operand is not a plain role name (an ``InverseRole``, or an OWL 2
|
|
1597
|
+
built-in property name, which the reader refuses in every position
|
|
1598
|
+
-- text it would not read back is never written), or the class
|
|
1599
|
+
expression of a ``Domain:`` / ``Range:`` axiom has one as the role
|
|
1600
|
+
of a restriction at any depth (a
|
|
1601
|
+
:class:`~unicode_logic_kit.dl.tableau.RoleExpressionError`, as for
|
|
1602
|
+
:func:`to_manchester`).
|
|
1603
|
+
"""
|
|
1604
|
+
# The tag is narrowed to `str` before any table lookup: the parameter is
|
|
1605
|
+
# `object` because the Domain:/Range: shapes carry a Concept, and a lookup
|
|
1606
|
+
# keyed on an un-narrowed `object` is exactly the kind of thing the pinned
|
|
1607
|
+
# mypy configuration refuses.
|
|
1608
|
+
tag = axiom[0] if axiom else None
|
|
1609
|
+
if not isinstance(tag, str):
|
|
1610
|
+
raise ValueError(
|
|
1611
|
+
f"role_axiom_to_manchester: the first element must be the shape's "
|
|
1612
|
+
f"TAG (a str), got {tag!r} in {axiom!r}")
|
|
1613
|
+
if len(axiom) == 3 and tag in _FILLER_FRAME_SPELLING:
|
|
1614
|
+
filler = axiom[2]
|
|
1615
|
+
if not isinstance(filler, Concept):
|
|
1616
|
+
raise ValueError(
|
|
1617
|
+
f"role_axiom_to_manchester: a {tag!r} axiom's third element is "
|
|
1618
|
+
f"its CLASS EXPRESSION, got {filler!r}")
|
|
1619
|
+
# The keyword keeps its colon here, unlike the role-to-role frames: the
|
|
1620
|
+
# W3C grammar's frame slot is `Domain: <description>` and the colon is
|
|
1621
|
+
# what tells a reader the right-hand side is a class expression (the
|
|
1622
|
+
# reader accepts it with or without, like every other frame keyword).
|
|
1623
|
+
role = _renderable_role(axiom[1], axiom)
|
|
1624
|
+
_reject_concept_roles_deep(filler, where="role_axiom_to_manchester")
|
|
1625
|
+
return f"{role} {_FILLER_FRAME_SPELLING[tag]}: {_render(filler)}"
|
|
1626
|
+
if len(axiom) == 3 and tag in _BINARY_FRAME_SPELLING:
|
|
1627
|
+
first = _renderable_role(axiom[1], axiom)
|
|
1628
|
+
second = _renderable_role(axiom[2], axiom)
|
|
1629
|
+
return f"{first} {_BINARY_FRAME_SPELLING[tag]} {second}"
|
|
1630
|
+
if len(axiom) == 2 and tag in _TAG_CHARACTERISTIC:
|
|
1631
|
+
role = _renderable_role(axiom[1], axiom)
|
|
1632
|
+
return f"{role} Characteristics: {_TAG_CHARACTERISTIC[tag]}"
|
|
1633
|
+
raise ValueError(
|
|
1634
|
+
f"role_axiom_to_manchester: expected one of "
|
|
1635
|
+
f"{sorted(_BINARY_FRAME_SPELLING)} with two roles, one of "
|
|
1636
|
+
f"{sorted(_FILLER_FRAME_SPELLING)} with a role and a concept, or one "
|
|
1637
|
+
f"of {sorted(_TAG_CHARACTERISTIC)} with one role, got {axiom!r}")
|