unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,818 @@
|
|
|
1
|
+
"""Datatypes, literals and data ranges: the data half of the OWL 2 DL layer.
|
|
2
|
+
|
|
3
|
+
OWL 2 has two disjoint domains: the OBJECT domain (individuals, ``Δ_I``) and
|
|
4
|
+
the DATA domain (``Δ_D``: numbers, strings, dates, … — the *data values*).
|
|
5
|
+
Classes and object properties live over the first, datatypes and data
|
|
6
|
+
properties relate the first to the second. This module holds the pieces of the
|
|
7
|
+
second that the rest of :mod:`unicode_logic_kit.dl` needs:
|
|
8
|
+
|
|
9
|
+
* :class:`Literal` — a typed literal such as ``"400"^^xsd:integer``, and the
|
|
10
|
+
term of the first-order image it becomes (:meth:`Literal.to_term`);
|
|
11
|
+
* the DATA-RANGE AST (:class:`Datatype`, :class:`DatatypeRestriction`,
|
|
12
|
+
:class:`DataOneOf`, :class:`DataComplementOf`, :class:`DataIntersectionOf`,
|
|
13
|
+
:class:`DataUnionOf`) and its first-order image
|
|
14
|
+
(:func:`datarange_to_fol`);
|
|
15
|
+
* the tables the two-sorted image reads (:func:`datatype_ancestors`,
|
|
16
|
+
:func:`datatype_family`) and the two reserved guard predicates
|
|
17
|
+
:data:`OWL_THING` / :data:`OWL_DATA`.
|
|
18
|
+
|
|
19
|
+
The image is GUARDED ONE-SORTED first-order logic
|
|
20
|
+
---------------------------------------------------
|
|
21
|
+
Not the kit's many-sorted logic (``SortedQuantifier``/``SortedConstant``): a
|
|
22
|
+
sorted constant must be lower-case, and every OEO individual and literal is
|
|
23
|
+
CamelCase. A datatype ``D`` becomes the unary predicate ``D(v)``; the object
|
|
24
|
+
domain and the data domain become the two reserved unary predicates
|
|
25
|
+
``OwlThing`` and ``OwlData``, and ``rdfs:Literal`` — the datatype whose value
|
|
26
|
+
space IS the data domain — maps onto ``OwlData`` exactly as ``owl:Thing``
|
|
27
|
+
already maps onto ``⊤``. The axioms that make the two predicates a
|
|
28
|
+
two-sorted theory (their disjointness, the typing of every property, the
|
|
29
|
+
datatype lattice, …) are SIDE axioms of the knowledge base, never conjuncts of
|
|
30
|
+
its formula: see :func:`unicode_logic_kit.dl.translate.data_sort_axioms`.
|
|
31
|
+
|
|
32
|
+
What a literal becomes
|
|
33
|
+
-----------------------
|
|
34
|
+
OWL's literal-to-value map is fixed, so a literal denotes ONE data value and
|
|
35
|
+
the image needs a TERM for it that is the same term for the same value and a
|
|
36
|
+
different one for a different value:
|
|
37
|
+
|
|
38
|
+
* a literal of the exact-number family (``xsd:integer`` and its subtypes,
|
|
39
|
+
``xsd:decimal``) becomes ``Number(value)`` — ``"1"^^xsd:integer`` and
|
|
40
|
+
``"1.0"^^xsd:decimal`` are the SAME data value in OWL 2 (the value spaces
|
|
41
|
+
overlap), so both become ``Number(1)``. A decimal that no kit ``Number``
|
|
42
|
+
can hold EXACTLY (more than 15 significant digits, where two different
|
|
43
|
+
decimals can be one float) is refused by name rather than read as another
|
|
44
|
+
number;
|
|
45
|
+
* a literal of ``xsd:float``/``xsd:double`` is REFUSED by name: their value
|
|
46
|
+
spaces are disjoint from the exact numbers (``-0``, ``NaN``, ``±INF``), so
|
|
47
|
+
``Number(1)`` would conflate two different values and any other term would
|
|
48
|
+
be a made-up one;
|
|
49
|
+
* ``xsd:normalizedString`` and ``xsd:token`` accept any string once XSD's
|
|
50
|
+
whitespace processing (§4.3.6) has run — *replace* (tab, line feed and
|
|
51
|
+
carriage return become a space) for the first, *collapse* (replace, then runs
|
|
52
|
+
of spaces become one and the ends are trimmed) for the second — so the literal
|
|
53
|
+
denotes the same VALUE as the ``xsd:string`` of the processed text and gets
|
|
54
|
+
that ``xsd:string`` term, typed by the datatype it was written with:
|
|
55
|
+
``" a b "^^xsd:token`` is ``"a b"^^xsd:string``. ``xsd:anyURI`` is a family
|
|
56
|
+
of its own, disjoint from the strings, whose term is the collapsed text: two
|
|
57
|
+
different texts are two different values. ``xsd:language``, ``xsd:Name``,
|
|
58
|
+
``xsd:NCName`` and ``xsd:NMTOKEN`` have lexical spaces (the XML name
|
|
59
|
+
productions) this kit does not validate, so it can neither say that a literal
|
|
60
|
+
of one denotes a value at all nor that it is the ``xsd:token`` of the same
|
|
61
|
+
text: they are REFUSED by name, like ``xsd:float``;
|
|
62
|
+
* every other literal becomes ``Constant('"abc"^^xsd:string')`` — its own
|
|
63
|
+
OWL text, kept verbatim, which is the convention :mod:`unicode_logic_kit.dl`
|
|
64
|
+
already follows for IRI-shaped names. The unicode syntax writes a constant of
|
|
65
|
+
any name in single quotes when its bare word would not read back as that
|
|
66
|
+
constant, so this one prints ``'"abc"^^xsd:string'`` and reads back through
|
|
67
|
+
``api.parse_any`` as the very constant (a single quote inside the lexical form
|
|
68
|
+
is escaped with a backslash between the quotes). What stays a documented limit
|
|
69
|
+
of the printed form, and a limit of RULE 5 (what the kit prints reads back
|
|
70
|
+
through ``api.parse_any``) that holds BY DESIGN for the data layer, is the name
|
|
71
|
+
of a built-in datatype: it is a PREDICATE, a predicate has no quoted form, and
|
|
72
|
+
so every image that names one (``xsd:integer(x0)``) prints text
|
|
73
|
+
``api.parse_any`` rejects, which ``tests/test_printed_text_reads_back.py``
|
|
74
|
+
carves out. The AST route (``api.prove`` over the nodes) is unaffected. To
|
|
75
|
+
print such an image as text the kit reads, rename its symbols with
|
|
76
|
+
:func:`unicode_logic_kit.fol.sanitize.sanitize_all` over the WHOLE premise list
|
|
77
|
+
with one shared mapping — ``sanitize_names`` applied to each formula with a
|
|
78
|
+
fresh mapping gives ``xsd:integer`` and a class called ``Xsdinteger`` the same
|
|
79
|
+
token, which is not injective.
|
|
80
|
+
|
|
81
|
+
The scope of facets
|
|
82
|
+
--------------------
|
|
83
|
+
The four ordering facets ``xsd:minInclusive`` / ``xsd:maxInclusive`` /
|
|
84
|
+
``xsd:minExclusive`` / ``xsd:maxExclusive`` on an exact-number base datatype
|
|
85
|
+
are the only facets with an image this kit can interpret: they become the
|
|
86
|
+
native comparison atoms ``≥ ≤ > <``. Every other facet (``xsd:pattern``, the
|
|
87
|
+
length facets, ``xsd:totalDigits``/``xsd:fractionDigits``, ``rdf:langRange``)
|
|
88
|
+
and an ordering facet on a non-numeric base (``xsd:dateTime``) is REFUSED by
|
|
89
|
+
name: its image would be an uninterpreted function or predicate that
|
|
90
|
+
constrains nothing while looking as though it did, which is exactly the silent
|
|
91
|
+
approximation this kit does not do.
|
|
92
|
+
"""
|
|
93
|
+
|
|
94
|
+
import re
|
|
95
|
+
from dataclasses import dataclass
|
|
96
|
+
from decimal import Decimal, InvalidOperation
|
|
97
|
+
from typing import Callable, Dict, FrozenSet, Iterable, Optional, Tuple
|
|
98
|
+
|
|
99
|
+
from ..fol._fol_nodes import _numeral_from_text
|
|
100
|
+
from ..fol.nodes import And, Atom, Constant, Node, Not, Number, Or
|
|
101
|
+
|
|
102
|
+
__all__ = [
|
|
103
|
+
"Literal", "DataRange", "Datatype", "DatatypeRestriction", "DataOneOf",
|
|
104
|
+
"DataComplementOf", "DataIntersectionOf", "DataUnionOf",
|
|
105
|
+
"UnsupportedDatatypeError", "datarange_to_fol", "OWL_THING", "OWL_DATA",
|
|
106
|
+
]
|
|
107
|
+
|
|
108
|
+
#: The reserved unary predicate of the OBJECT domain (``Δ_I``) in the
|
|
109
|
+
#: two-sorted image, and of the DATA domain (``Δ_D``). Fixed names, in the
|
|
110
|
+
#: way ``fol.qml`` fixes ``World``/``Object``: a knowledge base that uses
|
|
111
|
+
#: either as a class, role, property, individual or datatype name is refused
|
|
112
|
+
#: by name when the two-sorted axioms are asked for.
|
|
113
|
+
OWL_THING = "OwlThing"
|
|
114
|
+
OWL_DATA = "OwlData"
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
class UnsupportedDatatypeError(ValueError):
|
|
118
|
+
"""Raised for a datatype construct that is valid OWL 2 but has no faithful
|
|
119
|
+
first-order image in this kit: an out-of-scope facet, an ordering facet on
|
|
120
|
+
a non-numeric base, a literal whose value the kit cannot represent
|
|
121
|
+
exactly, an ill-typed literal, or a knowledge base that uses a name the
|
|
122
|
+
two-sorted image reserves.
|
|
123
|
+
|
|
124
|
+
Always a refusal BY NAME, with the construct, the reason and the spelling
|
|
125
|
+
to use instead — never an approximation.
|
|
126
|
+
"""
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
# --------------------------------------------------------------------------- #
|
|
130
|
+
# Names
|
|
131
|
+
# --------------------------------------------------------------------------- #
|
|
132
|
+
|
|
133
|
+
_NAMESPACES = (
|
|
134
|
+
("http://www.w3.org/2001/XMLSchema#", "xsd:"),
|
|
135
|
+
("http://www.w3.org/2000/01/rdf-schema#", "rdfs:"),
|
|
136
|
+
("http://www.w3.org/1999/02/22-rdf-syntax-ns#", "rdf:"),
|
|
137
|
+
("http://www.w3.org/2002/07/owl#", "owl:"),
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def canonical_datatype_name(name: str) -> str:
|
|
142
|
+
"""``name`` with a built-in namespace written as its usual prefix:
|
|
143
|
+
``http://www.w3.org/2001/XMLSchema#integer`` -- bracketed or not, as an IRI
|
|
144
|
+
is written in Manchester and Functional-Style Syntax -- and ``xsd:integer``
|
|
145
|
+
are one datatype, so they must be one NAME. Any other name is returned
|
|
146
|
+
unchanged (brackets and all: a user datatype's IRI is its own name).
|
|
147
|
+
"""
|
|
148
|
+
bare = name[1:-1] if name[:1] == "<" and name[-1:] == ">" else name
|
|
149
|
+
for namespace, prefix in _NAMESPACES:
|
|
150
|
+
if bare.startswith(namespace):
|
|
151
|
+
return prefix + bare[len(namespace):]
|
|
152
|
+
return name
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
_RDFS_LITERAL = "rdfs:Literal"
|
|
156
|
+
|
|
157
|
+
# Direct super-datatypes, from the OWL 2 datatype map (Structural Specification
|
|
158
|
+
# §4): the value space of each key is a SUBSET of its parents'.
|
|
159
|
+
_PARENTS: Dict[str, Tuple[str, ...]] = {
|
|
160
|
+
"owl:real": (),
|
|
161
|
+
"owl:rational": ("owl:real",),
|
|
162
|
+
"xsd:decimal": ("owl:rational",),
|
|
163
|
+
"xsd:integer": ("xsd:decimal",),
|
|
164
|
+
"xsd:nonNegativeInteger": ("xsd:integer",),
|
|
165
|
+
"xsd:positiveInteger": ("xsd:nonNegativeInteger",),
|
|
166
|
+
"xsd:nonPositiveInteger": ("xsd:integer",),
|
|
167
|
+
"xsd:negativeInteger": ("xsd:nonPositiveInteger",),
|
|
168
|
+
"xsd:long": ("xsd:integer",),
|
|
169
|
+
"xsd:int": ("xsd:long",),
|
|
170
|
+
"xsd:short": ("xsd:int",),
|
|
171
|
+
"xsd:byte": ("xsd:short",),
|
|
172
|
+
"xsd:unsignedLong": ("xsd:nonNegativeInteger",),
|
|
173
|
+
"xsd:unsignedInt": ("xsd:unsignedLong",),
|
|
174
|
+
"xsd:unsignedShort": ("xsd:unsignedInt",),
|
|
175
|
+
"xsd:unsignedByte": ("xsd:unsignedShort",),
|
|
176
|
+
"rdf:PlainLiteral": (),
|
|
177
|
+
"xsd:string": ("rdf:PlainLiteral",),
|
|
178
|
+
"xsd:normalizedString": ("xsd:string",),
|
|
179
|
+
"xsd:token": ("xsd:normalizedString",),
|
|
180
|
+
"xsd:language": ("xsd:token",),
|
|
181
|
+
"xsd:Name": ("xsd:token",),
|
|
182
|
+
"xsd:NCName": ("xsd:Name",),
|
|
183
|
+
"xsd:NMTOKEN": ("xsd:token",),
|
|
184
|
+
"xsd:double": (),
|
|
185
|
+
"xsd:float": (),
|
|
186
|
+
"xsd:boolean": (),
|
|
187
|
+
"xsd:hexBinary": (),
|
|
188
|
+
"xsd:base64Binary": (),
|
|
189
|
+
"xsd:anyURI": (),
|
|
190
|
+
"xsd:dateTime": (),
|
|
191
|
+
"xsd:dateTimeStamp": ("xsd:dateTime",),
|
|
192
|
+
"rdf:XMLLiteral": (),
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
# The datatypes whose value spaces are pairwise DISJOINT in OWL 2: one entry
|
|
196
|
+
# per "family root". Two built-in datatypes of different families share no
|
|
197
|
+
# value (OWL 2 §4: even xsd:float/xsd:double are disjoint from owl:real and
|
|
198
|
+
# from each other). Within a family nothing is claimed — `xsd:Name` and
|
|
199
|
+
# `xsd:NMTOKEN` overlap, `xsd:positiveInteger` and `xsd:negativeInteger` do not,
|
|
200
|
+
# and the lattice edges are the only within-family facts the image states.
|
|
201
|
+
_FAMILY_ROOTS = ("owl:real", "xsd:double", "xsd:float", "rdf:PlainLiteral",
|
|
202
|
+
"xsd:boolean", "xsd:hexBinary", "xsd:base64Binary",
|
|
203
|
+
"xsd:anyURI", "xsd:dateTime", "rdf:XMLLiteral")
|
|
204
|
+
|
|
205
|
+
#: Every datatype of the OWL 2 datatype map this module knows, plus ``rdfs:Literal``.
|
|
206
|
+
BUILTIN_DATATYPES: FrozenSet[str] = frozenset(_PARENTS) | {_RDFS_LITERAL}
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def datatype_ancestors(name: str) -> FrozenSet[str]:
|
|
210
|
+
"""Every built-in datatype that strictly CONTAINS ``name``'s value space
|
|
211
|
+
(transitively, ``rdfs:Literal`` excluded — it contains everything). Empty
|
|
212
|
+
for a name outside the OWL 2 datatype map: a user datatype's relation to
|
|
213
|
+
the lattice is whatever its own ``DatatypeDefinition`` says.
|
|
214
|
+
"""
|
|
215
|
+
name = canonical_datatype_name(name)
|
|
216
|
+
found = set()
|
|
217
|
+
stack = list(_PARENTS.get(name, ()))
|
|
218
|
+
while stack:
|
|
219
|
+
parent = stack.pop()
|
|
220
|
+
if parent not in found:
|
|
221
|
+
found.add(parent)
|
|
222
|
+
stack.extend(_PARENTS.get(parent, ()))
|
|
223
|
+
return frozenset(found)
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def datatype_family(name: str) -> Optional[str]:
|
|
227
|
+
"""The family root (one of :data:`_FAMILY_ROOTS`) a built-in datatype
|
|
228
|
+
belongs to, or ``None`` for ``rdfs:Literal`` and every name outside the
|
|
229
|
+
OWL 2 datatype map. Two datatypes with DIFFERENT non-``None`` families are
|
|
230
|
+
disjoint.
|
|
231
|
+
"""
|
|
232
|
+
name = canonical_datatype_name(name)
|
|
233
|
+
if name not in _PARENTS:
|
|
234
|
+
return None
|
|
235
|
+
for root in _FAMILY_ROOTS:
|
|
236
|
+
if name == root or root in datatype_ancestors(name):
|
|
237
|
+
return root
|
|
238
|
+
return None
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def is_builtin_datatype(name: str) -> bool:
|
|
242
|
+
"""True iff ``name`` is in the OWL 2 datatype map (or is ``rdfs:Literal``)."""
|
|
243
|
+
return canonical_datatype_name(name) in BUILTIN_DATATYPES
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
# --------------------------------------------------------------------------- #
|
|
247
|
+
# Literals
|
|
248
|
+
# --------------------------------------------------------------------------- #
|
|
249
|
+
|
|
250
|
+
# (min, max) of each integer-valued datatype; ``None`` is unbounded.
|
|
251
|
+
_INTEGER_TYPES: Dict[str, Tuple[Optional[int], Optional[int]]] = {
|
|
252
|
+
"xsd:integer": (None, None),
|
|
253
|
+
"xsd:nonNegativeInteger": (0, None),
|
|
254
|
+
"xsd:positiveInteger": (1, None),
|
|
255
|
+
"xsd:nonPositiveInteger": (None, 0),
|
|
256
|
+
"xsd:negativeInteger": (None, -1),
|
|
257
|
+
"xsd:long": (-2 ** 63, 2 ** 63 - 1),
|
|
258
|
+
"xsd:int": (-2 ** 31, 2 ** 31 - 1),
|
|
259
|
+
"xsd:short": (-2 ** 15, 2 ** 15 - 1),
|
|
260
|
+
"xsd:byte": (-2 ** 7, 2 ** 7 - 1),
|
|
261
|
+
"xsd:unsignedLong": (0, 2 ** 64 - 1),
|
|
262
|
+
"xsd:unsignedInt": (0, 2 ** 32 - 1),
|
|
263
|
+
"xsd:unsignedShort": (0, 2 ** 16 - 1),
|
|
264
|
+
"xsd:unsignedByte": (0, 2 ** 8 - 1),
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
#: The datatypes whose literals are EXACT numbers: the integer family and
|
|
268
|
+
#: ``xsd:decimal``. A literal of one of them becomes a ``Number``.
|
|
269
|
+
EXACT_NUMBER_DATATYPES: FrozenSet[str] = frozenset(_INTEGER_TYPES) | {"xsd:decimal"}
|
|
270
|
+
|
|
271
|
+
#: The datatypes an ordering facet may restrict: the exact numbers, plus the two
|
|
272
|
+
#: umbrella datatypes ``owl:real``/``owl:rational`` (which have no literals of
|
|
273
|
+
#: their own but are legal bases).
|
|
274
|
+
NUMERIC_BASES: FrozenSet[str] = EXACT_NUMBER_DATATYPES | {"owl:real", "owl:rational"}
|
|
275
|
+
|
|
276
|
+
_INTEGER_LEXICAL = re.compile(r"[+-]?[0-9]+")
|
|
277
|
+
_DECIMAL_LEXICAL = re.compile(r"[+-]?(?:[0-9]+(?:\.[0-9]*)?|\.[0-9]+)")
|
|
278
|
+
_FLOATING = frozenset({"xsd:float", "xsd:double"})
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def _whitespace_replace(text: str) -> str:
|
|
282
|
+
"""XSD 1.1 §4.3.6 ``whiteSpace = replace``: tab, line feed and carriage
|
|
283
|
+
return each become a space. (Exactly those three: Python's ``str.split`` and
|
|
284
|
+
``strip`` also treat form feed, NBSP and the Unicode spaces as white space,
|
|
285
|
+
which XSD does not.)"""
|
|
286
|
+
return text.replace("\t", " ").replace("\n", " ").replace("\r", " ")
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def _whitespace_collapse(text: str) -> str:
|
|
290
|
+
"""XSD 1.1 §4.3.6 ``whiteSpace = collapse``: *replace*, then runs of spaces
|
|
291
|
+
become one space and a leading or trailing space is removed."""
|
|
292
|
+
return re.sub(" +", " ", _whitespace_replace(text)).strip(" ")
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
#: The datatypes whose lexical form is whitespace-processed before it denotes a
|
|
296
|
+
#: value, with the processing. ``xsd:string`` (and ``rdf:PlainLiteral``) PRESERVE.
|
|
297
|
+
_WHITESPACE_PROCESSING: Dict[str, Callable[[str], str]] = {
|
|
298
|
+
"xsd:normalizedString": _whitespace_replace,
|
|
299
|
+
"xsd:token": _whitespace_collapse,
|
|
300
|
+
"xsd:anyURI": _whitespace_collapse,
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
#: ``xsd:string`` subtypes with no lexical constraint beyond the whitespace
|
|
304
|
+
#: processing: their literal denotes the ``xsd:string`` value of the processed text.
|
|
305
|
+
_STRING_VALUED = frozenset({"xsd:normalizedString", "xsd:token"})
|
|
306
|
+
|
|
307
|
+
#: Datatypes whose lexical space is an XML name production this kit does not
|
|
308
|
+
#: validate: a literal of one has no term (see :meth:`Literal.to_term`).
|
|
309
|
+
_UNVALIDATED_LEXICAL = frozenset({"xsd:language", "xsd:Name", "xsd:NCName",
|
|
310
|
+
"xsd:NMTOKEN"})
|
|
311
|
+
_BOOLEAN_CANONICAL = {"true": "true", "1": "true", "false": "false", "0": "false"}
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def _escape(lexical: str) -> str:
|
|
315
|
+
"""The OWL 2 Functional-Style string escape: a backslash before ``\\`` and ``"``."""
|
|
316
|
+
return lexical.replace("\\", "\\\\").replace('"', '\\"')
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
@dataclass(frozen=True)
|
|
320
|
+
class Literal:
|
|
321
|
+
"""A typed literal ``"lexical"^^datatype`` (or a language-tagged string
|
|
322
|
+
``"lexical"@language``), OWL's ``DataPropertyAssertion`` value and the
|
|
323
|
+
value operand of ``DataHasValue``/``DataOneOf``/a facet.
|
|
324
|
+
|
|
325
|
+
Args:
|
|
326
|
+
lexical: The lexical form, as written (``"400"``).
|
|
327
|
+
datatype: The datatype name; a built-in namespace is canonicalised to
|
|
328
|
+
its prefix, so ``<http://www.w3.org/2001/XMLSchema#integer>`` and
|
|
329
|
+
``xsd:integer`` are one datatype. Defaults to ``xsd:string``.
|
|
330
|
+
language: A language tag, which makes the literal an ``rdf:PlainLiteral``
|
|
331
|
+
(the only datatype that has one); stored lower-case, as language
|
|
332
|
+
tags are case-insensitive.
|
|
333
|
+
|
|
334
|
+
Raises:
|
|
335
|
+
UnsupportedDatatypeError: the lexical form is ILL-TYPED for an
|
|
336
|
+
exact-number datatype or for ``xsd:boolean`` (``"abc"^^xsd:integer``
|
|
337
|
+
has no data value — an ontology containing it is inconsistent in
|
|
338
|
+
OWL 2 and the kit does not guess), or a language tag is given on a
|
|
339
|
+
datatype that has none. Other datatypes are not checked: only for
|
|
340
|
+
the exact numbers, ``xsd:boolean`` and strings does this kit
|
|
341
|
+
interpret the lexical form at all.
|
|
342
|
+
"""
|
|
343
|
+
|
|
344
|
+
lexical: str
|
|
345
|
+
datatype: str = "xsd:string"
|
|
346
|
+
language: Optional[str] = None
|
|
347
|
+
|
|
348
|
+
def __post_init__(self):
|
|
349
|
+
datatype = canonical_datatype_name(self.datatype)
|
|
350
|
+
if self.language is not None:
|
|
351
|
+
if datatype not in ("xsd:string", "rdf:PlainLiteral"):
|
|
352
|
+
raise UnsupportedDatatypeError(
|
|
353
|
+
f"dl.Literal: a language tag ({self.language!r}) belongs to "
|
|
354
|
+
f"rdf:PlainLiteral, not to {datatype!r}. Write the literal "
|
|
355
|
+
f"as \"{self.lexical}\"@{self.language} or drop the tag.")
|
|
356
|
+
datatype = "rdf:PlainLiteral"
|
|
357
|
+
object.__setattr__(self, "language", self.language.lower())
|
|
358
|
+
elif datatype == "rdf:PlainLiteral":
|
|
359
|
+
raise UnsupportedDatatypeError(
|
|
360
|
+
f"dl.Literal: a rdf:PlainLiteral literal {self.lexical!r} without "
|
|
361
|
+
f"a language tag is the xsd:string {self.lexical!r}; write it as "
|
|
362
|
+
f"an xsd:string, or give a language tag.")
|
|
363
|
+
object.__setattr__(self, "datatype", datatype)
|
|
364
|
+
self._check_well_typed()
|
|
365
|
+
|
|
366
|
+
def _check_well_typed(self) -> None:
|
|
367
|
+
text = self.lexical.strip()
|
|
368
|
+
if self.datatype in _INTEGER_TYPES:
|
|
369
|
+
low, high = _INTEGER_TYPES[self.datatype]
|
|
370
|
+
if not _INTEGER_LEXICAL.fullmatch(text) or not (
|
|
371
|
+
(low is None or int(text) >= low)
|
|
372
|
+
and (high is None or int(text) <= high)):
|
|
373
|
+
raise UnsupportedDatatypeError(
|
|
374
|
+
f"dl.Literal: {self.lexical!r} is not a well-typed "
|
|
375
|
+
f"{self.datatype} literal (an ill-typed literal denotes no "
|
|
376
|
+
f"data value, and an ontology containing one is inconsistent "
|
|
377
|
+
f"in OWL 2 — this kit refuses it rather than guess).")
|
|
378
|
+
elif self.datatype == "xsd:decimal":
|
|
379
|
+
if not _DECIMAL_LEXICAL.fullmatch(text):
|
|
380
|
+
raise UnsupportedDatatypeError(
|
|
381
|
+
f"dl.Literal: {self.lexical!r} is not a well-typed "
|
|
382
|
+
f"xsd:decimal literal (digits with an optional sign and "
|
|
383
|
+
f"decimal point; no exponent).")
|
|
384
|
+
elif self.datatype == "xsd:boolean":
|
|
385
|
+
if text not in _BOOLEAN_CANONICAL:
|
|
386
|
+
raise UnsupportedDatatypeError(
|
|
387
|
+
f"dl.Literal: {self.lexical!r} is not a well-typed "
|
|
388
|
+
f"xsd:boolean literal (one of true, false, 1, 0).")
|
|
389
|
+
|
|
390
|
+
# -- text ------------------------------------------------------------- #
|
|
391
|
+
|
|
392
|
+
def canonical_lexical(self) -> str:
|
|
393
|
+
"""The lexical form the TERM is named after: ``xsd:boolean`` is folded
|
|
394
|
+
onto ``true``/``false`` (``"1"`` and ``"true"`` are one value), the three
|
|
395
|
+
datatypes whose whitespace facet is not *preserve* are folded onto the
|
|
396
|
+
text XSD's whitespace processing makes of the lexical form
|
|
397
|
+
(``xsd:normalizedString`` *replace*, ``xsd:token`` and ``xsd:anyURI``
|
|
398
|
+
*collapse*) and every other datatype keeps the lexical form it was
|
|
399
|
+
written with.
|
|
400
|
+
"""
|
|
401
|
+
if self.datatype == "xsd:boolean":
|
|
402
|
+
return _BOOLEAN_CANONICAL[self.lexical.strip()]
|
|
403
|
+
process = _WHITESPACE_PROCESSING.get(self.datatype)
|
|
404
|
+
if process is not None:
|
|
405
|
+
return process(self.lexical)
|
|
406
|
+
return self.lexical
|
|
407
|
+
|
|
408
|
+
def to_unicode(self) -> str:
|
|
409
|
+
"""The OWL 2 text of the literal: ``"400"^^xsd:integer`` or
|
|
410
|
+
``"abc"@en``. A name written back in this spelling reads back as this
|
|
411
|
+
literal through the OWL parsers.
|
|
412
|
+
"""
|
|
413
|
+
if self.language is not None:
|
|
414
|
+
return f'"{_escape(self.lexical)}"@{self.language}'
|
|
415
|
+
return f'"{_escape(self.lexical)}"^^{self.datatype}'
|
|
416
|
+
|
|
417
|
+
def __str__(self) -> str:
|
|
418
|
+
return self.to_unicode()
|
|
419
|
+
|
|
420
|
+
# -- value / term ----------------------------------------------------- #
|
|
421
|
+
|
|
422
|
+
def exact_number(self) -> Optional[Decimal]:
|
|
423
|
+
"""The literal's value as an exact :class:`~decimal.Decimal` if its
|
|
424
|
+
datatype is in the exact-number family, else ``None``."""
|
|
425
|
+
if self.datatype not in EXACT_NUMBER_DATATYPES:
|
|
426
|
+
return None
|
|
427
|
+
try:
|
|
428
|
+
return Decimal(self.lexical.strip())
|
|
429
|
+
except InvalidOperation: # pragma: no cover - checked at construction
|
|
430
|
+
return None
|
|
431
|
+
|
|
432
|
+
def to_term(self) -> Node:
|
|
433
|
+
"""The first-order TERM of the literal's data value.
|
|
434
|
+
|
|
435
|
+
An exact number becomes ``Number`` (an ``int`` when integral, else a
|
|
436
|
+
``float`` — and only when the decimal has at most 15 significant digits,
|
|
437
|
+
the rule of every numeral reader of the kit: see
|
|
438
|
+
:func:`~unicode_logic_kit.fol._fol_nodes._numeral_from_text`).
|
|
439
|
+
``xsd:float``/``xsd:double``, a decimal no ``Number`` can
|
|
440
|
+
hold exactly and the datatypes whose lexical space this kit does not
|
|
441
|
+
validate (``xsd:language``, ``xsd:Name``, ``xsd:NCName``,
|
|
442
|
+
``xsd:NMTOKEN``) are REFUSED. ``xsd:normalizedString`` and ``xsd:token``
|
|
443
|
+
become the ``xsd:string`` term of their whitespace-processed text.
|
|
444
|
+
Everything else becomes a ``Constant`` named by the literal's own OWL
|
|
445
|
+
text (see the module docstring).
|
|
446
|
+
|
|
447
|
+
Raises:
|
|
448
|
+
UnsupportedDatatypeError: see above.
|
|
449
|
+
"""
|
|
450
|
+
if self.datatype in _FLOATING:
|
|
451
|
+
raise UnsupportedDatatypeError(
|
|
452
|
+
f"dl.Literal.to_term: the literal {self.to_unicode()} has no "
|
|
453
|
+
f"first-order image in this kit. The value space of "
|
|
454
|
+
f"{self.datatype} is disjoint from the exact numbers in OWL 2 "
|
|
455
|
+
f"(it has -0, NaN and ±INF, and 1.0 is not the integer 1), so "
|
|
456
|
+
f"the term Number(1.0) would conflate two different values and "
|
|
457
|
+
f"any other term would be invented. Use xsd:decimal or "
|
|
458
|
+
f"xsd:integer, or state the constraint outside the datatype.")
|
|
459
|
+
if self.datatype in _UNVALIDATED_LEXICAL:
|
|
460
|
+
raise UnsupportedDatatypeError(
|
|
461
|
+
f"dl.Literal.to_term: the literal {self.to_unicode()} has no "
|
|
462
|
+
f"first-order image in this kit. The lexical space of "
|
|
463
|
+
f"{self.datatype} is an XML name production this kit does not "
|
|
464
|
+
f"validate, so it cannot say whether the literal denotes a "
|
|
465
|
+
f"value at all, nor that it is the xsd:token / xsd:string value "
|
|
466
|
+
f"of the same text, and a term of its own would state a "
|
|
467
|
+
f"distinctness or an identity the value space does not "
|
|
468
|
+
f"guarantee. Write the value as an xsd:token or xsd:string "
|
|
469
|
+
f"literal, or leave it out.")
|
|
470
|
+
if self.datatype in _STRING_VALUED:
|
|
471
|
+
# The same VALUE as the xsd:string of the processed text, so the
|
|
472
|
+
# same term: the typing by the written datatype is a separate fact.
|
|
473
|
+
return Constant(Literal(self.canonical_lexical(), "xsd:string")
|
|
474
|
+
.to_unicode())
|
|
475
|
+
value = self.exact_number()
|
|
476
|
+
if value is None:
|
|
477
|
+
return Constant(Literal(self.canonical_lexical(), self.datatype,
|
|
478
|
+
self.language).to_unicode())
|
|
479
|
+
if value == value.to_integral_value():
|
|
480
|
+
return Number(int(value))
|
|
481
|
+
try:
|
|
482
|
+
return Number(_numeral_from_text(format(value, "f")))
|
|
483
|
+
except ValueError as exc:
|
|
484
|
+
raise UnsupportedDatatypeError(
|
|
485
|
+
f"dl.Literal.to_term: the decimal {self.to_unicode()} cannot be "
|
|
486
|
+
f"held exactly by a kit Number, and reading it as the nearest "
|
|
487
|
+
f"float would let two different data values collapse into one "
|
|
488
|
+
f"term: {exc}.") from None
|
|
489
|
+
|
|
490
|
+
|
|
491
|
+
# --------------------------------------------------------------------------- #
|
|
492
|
+
# Data ranges
|
|
493
|
+
# --------------------------------------------------------------------------- #
|
|
494
|
+
|
|
495
|
+
class DataRange:
|
|
496
|
+
"""Base class for OWL 2 data ranges (see the module docstring)."""
|
|
497
|
+
|
|
498
|
+
def to_unicode(self) -> str:
|
|
499
|
+
"""The data range in the kit's display notation (``xsd:integer[≥ 5]``,
|
|
500
|
+
``{1, 2}``, ``¬D``, ``D ⊓ E``, ``D ⊔ E``)."""
|
|
501
|
+
return _render_range(self)
|
|
502
|
+
|
|
503
|
+
def __str__(self) -> str:
|
|
504
|
+
return self.to_unicode()
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
@dataclass(frozen=True)
|
|
508
|
+
class Datatype(DataRange):
|
|
509
|
+
"""A named datatype (``xsd:integer``, ``rdfs:Literal``, a user datatype
|
|
510
|
+
defined by ``DatatypeDefinition``). A built-in namespace is canonicalised
|
|
511
|
+
to its prefix."""
|
|
512
|
+
|
|
513
|
+
name: str
|
|
514
|
+
|
|
515
|
+
def __post_init__(self):
|
|
516
|
+
object.__setattr__(self, "name", canonical_datatype_name(self.name))
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
#: The ordering facets, in canonical spelling, with the comparison atom each
|
|
520
|
+
#: becomes: ``value ≥ bound`` for ``xsd:minInclusive`` and so on.
|
|
521
|
+
_ORDER_FACETS = {
|
|
522
|
+
"xsd:minInclusive": "≥",
|
|
523
|
+
"xsd:maxInclusive": "≤",
|
|
524
|
+
"xsd:minExclusive": ">",
|
|
525
|
+
"xsd:maxExclusive": "<",
|
|
526
|
+
}
|
|
527
|
+
|
|
528
|
+
#: The facets this kit translates (see the module docstring's "The scope of facets").
|
|
529
|
+
SUPPORTED_FACETS: Tuple[str, ...] = tuple(_ORDER_FACETS)
|
|
530
|
+
|
|
531
|
+
_NO_IMAGE_FACETS = {
|
|
532
|
+
"xsd:pattern": "regular-expression constraints over a literal's lexical "
|
|
533
|
+
"space are not first-order, and no backend here interprets them",
|
|
534
|
+
"xsd:length": "string length is an uninterpreted function on every backend "
|
|
535
|
+
"here, so the axiom would constrain nothing while looking as "
|
|
536
|
+
"though it did",
|
|
537
|
+
"xsd:minLength": "string length is an uninterpreted function on every "
|
|
538
|
+
"backend here, so the axiom would constrain nothing while "
|
|
539
|
+
"looking as though it did",
|
|
540
|
+
"xsd:maxLength": "string length is an uninterpreted function on every "
|
|
541
|
+
"backend here, so the axiom would constrain nothing while "
|
|
542
|
+
"looking as though it did",
|
|
543
|
+
"xsd:totalDigits": "digit counts are not first-order over the kit's numbers",
|
|
544
|
+
"xsd:fractionDigits": "digit counts are not first-order over the kit's numbers",
|
|
545
|
+
"rdf:langRange": "language-range matching is not first-order over the "
|
|
546
|
+
"kit's strings",
|
|
547
|
+
}
|
|
548
|
+
|
|
549
|
+
|
|
550
|
+
def _check_facet(base: str, facet: str, bound: "Literal") -> str:
|
|
551
|
+
"""Validate one facet of a restriction of ``base``; return its canonical name."""
|
|
552
|
+
facet = canonical_datatype_name(facet)
|
|
553
|
+
if facet not in _ORDER_FACETS:
|
|
554
|
+
why = _NO_IMAGE_FACETS.get(facet, "it is not one of the four ordering facets")
|
|
555
|
+
raise UnsupportedDatatypeError(
|
|
556
|
+
f"dl.datatypes: the facet {facet!r} has no first-order image in this "
|
|
557
|
+
f"kit — {why}. Supported facets are {', '.join(SUPPORTED_FACETS)} "
|
|
558
|
+
f"on an exact-number base datatype; drop the facet, or state the "
|
|
559
|
+
f"constraint outside the datatype.")
|
|
560
|
+
if base not in NUMERIC_BASES:
|
|
561
|
+
raise UnsupportedDatatypeError(
|
|
562
|
+
f"dl.datatypes: the facet {facet!r} is supported only on an "
|
|
563
|
+
f"exact-number base datatype (xsd:integer and its subtypes, "
|
|
564
|
+
f"xsd:decimal), not on {base!r} — this kit's ≤/≥ atoms carry no "
|
|
565
|
+
f"theory for it, so the image would be an uninterpreted predicate "
|
|
566
|
+
f"over uninterpreted constants and would constrain nothing. Use "
|
|
567
|
+
f"xsd:decimal or xsd:integer, or state the bound outside the "
|
|
568
|
+
f"datatype.")
|
|
569
|
+
if bound.exact_number() is None:
|
|
570
|
+
raise UnsupportedDatatypeError(
|
|
571
|
+
f"dl.datatypes: the bound {bound.to_unicode()} of the facet "
|
|
572
|
+
f"{facet!r} is not an exact number (xsd:integer family or "
|
|
573
|
+
f"xsd:decimal literal), so it has no ordering term in this kit.")
|
|
574
|
+
return facet
|
|
575
|
+
|
|
576
|
+
|
|
577
|
+
@dataclass(frozen=True)
|
|
578
|
+
class DatatypeRestriction(DataRange):
|
|
579
|
+
"""``DatatypeRestriction(base facet₁ literal₁ …)``: the values of ``base``
|
|
580
|
+
that satisfy every facet. ``facets`` is a tuple of ``(facet name, bound
|
|
581
|
+
literal)``; it is validated against the supported scope on construction.
|
|
582
|
+
|
|
583
|
+
Raises:
|
|
584
|
+
UnsupportedDatatypeError: an out-of-scope facet, or an ordering facet
|
|
585
|
+
on a non-numeric base (see the module docstring).
|
|
586
|
+
"""
|
|
587
|
+
|
|
588
|
+
base: Datatype
|
|
589
|
+
facets: Tuple[Tuple[str, Literal], ...]
|
|
590
|
+
|
|
591
|
+
def __post_init__(self):
|
|
592
|
+
base = self.base if isinstance(self.base, Datatype) else Datatype(self.base)
|
|
593
|
+
object.__setattr__(self, "base", base)
|
|
594
|
+
facets = tuple((facet, bound) for facet, bound in self.facets)
|
|
595
|
+
if not facets:
|
|
596
|
+
raise UnsupportedDatatypeError(
|
|
597
|
+
"dl.DatatypeRestriction: a restriction needs at least one facet "
|
|
598
|
+
"(OWL 2's grammar is DatatypeRestriction(DT (F lt)+)); a datatype "
|
|
599
|
+
"with no facet is just the datatype.")
|
|
600
|
+
checked = tuple((_check_facet(base.name, facet, bound), bound)
|
|
601
|
+
for facet, bound in facets)
|
|
602
|
+
object.__setattr__(self, "facets", checked)
|
|
603
|
+
|
|
604
|
+
|
|
605
|
+
@dataclass(frozen=True)
|
|
606
|
+
class DataOneOf(DataRange):
|
|
607
|
+
"""``DataOneOf(lt₁ … ltₙ)``: exactly the listed data values."""
|
|
608
|
+
|
|
609
|
+
values: Tuple[Literal, ...]
|
|
610
|
+
|
|
611
|
+
def __post_init__(self):
|
|
612
|
+
object.__setattr__(self, "values", tuple(self.values))
|
|
613
|
+
if not self.values:
|
|
614
|
+
raise UnsupportedDatatypeError(
|
|
615
|
+
"dl.DataOneOf: needs at least one literal (OWL 2's grammar is "
|
|
616
|
+
"DataOneOf(lt+)).")
|
|
617
|
+
|
|
618
|
+
|
|
619
|
+
@dataclass(frozen=True)
|
|
620
|
+
class DataComplementOf(DataRange):
|
|
621
|
+
"""``DataComplementOf(DR)``: the data values NOT in ``DR`` — the
|
|
622
|
+
complement within the data domain, not within the whole universe."""
|
|
623
|
+
|
|
624
|
+
datarange: DataRange
|
|
625
|
+
|
|
626
|
+
|
|
627
|
+
@dataclass(frozen=True)
|
|
628
|
+
class DataIntersectionOf(DataRange):
|
|
629
|
+
"""``DataIntersectionOf(DR₁ … DRₙ)``, ``n ≥ 2``."""
|
|
630
|
+
|
|
631
|
+
ranges: Tuple[DataRange, ...]
|
|
632
|
+
|
|
633
|
+
def __post_init__(self):
|
|
634
|
+
object.__setattr__(self, "ranges", tuple(self.ranges))
|
|
635
|
+
if len(self.ranges) < 2:
|
|
636
|
+
raise UnsupportedDatatypeError(
|
|
637
|
+
"dl.DataIntersectionOf: needs at least two data ranges.")
|
|
638
|
+
|
|
639
|
+
|
|
640
|
+
@dataclass(frozen=True)
|
|
641
|
+
class DataUnionOf(DataRange):
|
|
642
|
+
"""``DataUnionOf(DR₁ … DRₙ)``, ``n ≥ 2``."""
|
|
643
|
+
|
|
644
|
+
ranges: Tuple[DataRange, ...]
|
|
645
|
+
|
|
646
|
+
def __post_init__(self):
|
|
647
|
+
object.__setattr__(self, "ranges", tuple(self.ranges))
|
|
648
|
+
if len(self.ranges) < 2:
|
|
649
|
+
raise UnsupportedDatatypeError(
|
|
650
|
+
"dl.DataUnionOf: needs at least two data ranges.")
|
|
651
|
+
|
|
652
|
+
|
|
653
|
+
# --------------------------------------------------------------------------- #
|
|
654
|
+
# Display
|
|
655
|
+
# --------------------------------------------------------------------------- #
|
|
656
|
+
|
|
657
|
+
_FACET_SYMBOL = {facet: symbol for facet, symbol in _ORDER_FACETS.items()}
|
|
658
|
+
|
|
659
|
+
|
|
660
|
+
def _render_range(dr: DataRange) -> str:
|
|
661
|
+
if isinstance(dr, Datatype):
|
|
662
|
+
return dr.name
|
|
663
|
+
if isinstance(dr, DatatypeRestriction):
|
|
664
|
+
facets = ", ".join(f"{_FACET_SYMBOL[f]} {b.canonical_lexical()}"
|
|
665
|
+
for f, b in dr.facets)
|
|
666
|
+
return f"{dr.base.name}[{facets}]"
|
|
667
|
+
if isinstance(dr, DataOneOf):
|
|
668
|
+
return "{" + ", ".join(v.to_unicode() for v in dr.values) + "}"
|
|
669
|
+
if isinstance(dr, DataComplementOf):
|
|
670
|
+
return "¬" + _paren_range(dr.datarange)
|
|
671
|
+
if isinstance(dr, DataIntersectionOf):
|
|
672
|
+
return " ⊓ ".join(_paren_range(r) for r in dr.ranges)
|
|
673
|
+
if isinstance(dr, DataUnionOf):
|
|
674
|
+
return " ⊔ ".join(_paren_range(r) for r in dr.ranges)
|
|
675
|
+
raise TypeError(f"render: unsupported data range {type(dr).__name__}")
|
|
676
|
+
|
|
677
|
+
|
|
678
|
+
def _paren_range(dr: DataRange) -> str:
|
|
679
|
+
inner = _render_range(dr)
|
|
680
|
+
if isinstance(dr, (DataIntersectionOf, DataUnionOf)):
|
|
681
|
+
return f"({inner})"
|
|
682
|
+
return inner
|
|
683
|
+
|
|
684
|
+
|
|
685
|
+
# --------------------------------------------------------------------------- #
|
|
686
|
+
# The first-order image of a data range
|
|
687
|
+
# --------------------------------------------------------------------------- #
|
|
688
|
+
|
|
689
|
+
def _fold(ctor, parts: Iterable[Node]) -> Node:
|
|
690
|
+
parts = list(parts)
|
|
691
|
+
acc = parts[0]
|
|
692
|
+
for part in parts[1:]:
|
|
693
|
+
acc = ctor(acc, part)
|
|
694
|
+
return acc
|
|
695
|
+
|
|
696
|
+
|
|
697
|
+
def datarange_to_fol(datarange: DataRange, term: Node) -> Node:
|
|
698
|
+
"""The first-order image ``δ(DR, t)`` of a data range, with ``term`` (a
|
|
699
|
+
``Variable``, or any other term) standing for the data value under
|
|
700
|
+
discussion. OWL 2 direct semantics (Structural Specification §7):
|
|
701
|
+
|
|
702
|
+
* ``Datatype(D)`` ↦ ``D(t)``; ``rdfs:Literal`` — whose value space is the
|
|
703
|
+
whole data domain — ↦ ``OwlData(t)``;
|
|
704
|
+
* ``DatatypeRestriction(D f₁ v₁ … fₙ vₙ)`` ↦ ``D(t) ∧ t ⋈₁ v₁ ∧ … ∧ t ⋈ₙ vₙ``
|
|
705
|
+
(``(D)^DT ∩ ⋂ (fᵢ, vᵢ)^F`` is the conjunction of the base guard with
|
|
706
|
+
one comparison atom per facet);
|
|
707
|
+
* ``DataOneOf(l₁ … lₙ)`` ↦ ``t = l₁ ∨ … ∨ t = lₙ``;
|
|
708
|
+
* ``DataComplementOf(DR)`` ↦ ``OwlData(t) ∧ ¬δ(DR, t)`` — the complement
|
|
709
|
+
is taken WITHIN the data domain (``Δ_D \\ DR^DT``), so the ``OwlData``
|
|
710
|
+
conjunct is load-bearing: without it the image would also admit
|
|
711
|
+
individuals;
|
|
712
|
+
* ``DataIntersectionOf`` ↦ ``∧``, ``DataUnionOf`` ↦ ``∨``.
|
|
713
|
+
|
|
714
|
+
Raises:
|
|
715
|
+
UnsupportedDatatypeError: a literal with no term (``xsd:double``, an
|
|
716
|
+
inexact decimal).
|
|
717
|
+
"""
|
|
718
|
+
if isinstance(datarange, Datatype):
|
|
719
|
+
if datarange.name == _RDFS_LITERAL:
|
|
720
|
+
return Atom(OWL_DATA, (term,))
|
|
721
|
+
return Atom(datarange.name, (term,))
|
|
722
|
+
if isinstance(datarange, DatatypeRestriction):
|
|
723
|
+
parts = [datarange_to_fol(datarange.base, term)]
|
|
724
|
+
for facet, bound in datarange.facets:
|
|
725
|
+
parts.append(Atom(_ORDER_FACETS[facet], (term, bound.to_term())))
|
|
726
|
+
return _fold(And, parts)
|
|
727
|
+
if isinstance(datarange, DataOneOf):
|
|
728
|
+
return _fold(Or, [Atom("=", (term, value.to_term()))
|
|
729
|
+
for value in datarange.values])
|
|
730
|
+
if isinstance(datarange, DataComplementOf):
|
|
731
|
+
return And(Atom(OWL_DATA, (term,)),
|
|
732
|
+
Not(datarange_to_fol(datarange.datarange, term)))
|
|
733
|
+
if isinstance(datarange, DataIntersectionOf):
|
|
734
|
+
return _fold(And, [datarange_to_fol(r, term) for r in datarange.ranges])
|
|
735
|
+
if isinstance(datarange, DataUnionOf):
|
|
736
|
+
return _fold(Or, [datarange_to_fol(r, term) for r in datarange.ranges])
|
|
737
|
+
raise TypeError(f"datarange_to_fol: unsupported data range {type(datarange).__name__}")
|
|
738
|
+
|
|
739
|
+
|
|
740
|
+
# --------------------------------------------------------------------------- #
|
|
741
|
+
# Walking a data range
|
|
742
|
+
# --------------------------------------------------------------------------- #
|
|
743
|
+
|
|
744
|
+
def datarange_literals(datarange: DataRange) -> Tuple[Literal, ...]:
|
|
745
|
+
"""Every :class:`Literal` occurring in ``datarange`` (facet bounds and
|
|
746
|
+
one-of values), in order of occurrence."""
|
|
747
|
+
if isinstance(datarange, Datatype):
|
|
748
|
+
return ()
|
|
749
|
+
if isinstance(datarange, DatatypeRestriction):
|
|
750
|
+
return tuple(bound for _, bound in datarange.facets)
|
|
751
|
+
if isinstance(datarange, DataOneOf):
|
|
752
|
+
return datarange.values
|
|
753
|
+
if isinstance(datarange, DataComplementOf):
|
|
754
|
+
return datarange_literals(datarange.datarange)
|
|
755
|
+
if isinstance(datarange, (DataIntersectionOf, DataUnionOf)):
|
|
756
|
+
return tuple(lit for r in datarange.ranges for lit in datarange_literals(r))
|
|
757
|
+
raise TypeError(f"datarange_literals: unsupported data range {type(datarange).__name__}")
|
|
758
|
+
|
|
759
|
+
|
|
760
|
+
def datarange_datatypes(datarange: DataRange) -> Tuple[str, ...]:
|
|
761
|
+
"""Every datatype NAME occurring in ``datarange`` (a base, or a plain
|
|
762
|
+
datatype), in order of occurrence — not the literals' datatypes, see
|
|
763
|
+
:func:`datarange_literals`."""
|
|
764
|
+
if isinstance(datarange, Datatype):
|
|
765
|
+
return (datarange.name,)
|
|
766
|
+
if isinstance(datarange, DatatypeRestriction):
|
|
767
|
+
return (datarange.base.name,)
|
|
768
|
+
if isinstance(datarange, DataOneOf):
|
|
769
|
+
return ()
|
|
770
|
+
if isinstance(datarange, DataComplementOf):
|
|
771
|
+
return datarange_datatypes(datarange.datarange)
|
|
772
|
+
if isinstance(datarange, (DataIntersectionOf, DataUnionOf)):
|
|
773
|
+
return tuple(name for r in datarange.ranges for name in datarange_datatypes(r))
|
|
774
|
+
raise TypeError(f"datarange_datatypes: unsupported data range {type(datarange).__name__}")
|
|
775
|
+
|
|
776
|
+
|
|
777
|
+
#: ``(callable)`` alias used by the OWL writers: maps a datatype NAME to the
|
|
778
|
+
#: text it is written as (``<IRI>`` brackets, a synthetic name, …).
|
|
779
|
+
NameRenderer = Callable[[str], str]
|
|
780
|
+
|
|
781
|
+
|
|
782
|
+
def render_literal_fs(literal: Literal, name: NameRenderer = lambda n: n) -> str:
|
|
783
|
+
"""The OWL 2 Functional-Style text of ``literal`` (``"400"^^xsd:integer``);
|
|
784
|
+
``name`` renders its DATATYPE name, exactly as it does the datatype names of
|
|
785
|
+
:func:`render_datarange_fs` (default: verbatim).
|
|
786
|
+
|
|
787
|
+
A writer that renders datatype names through a renderer of its own (an IRI in
|
|
788
|
+
angle brackets, a synthetic ``:T1`` token) must render a literal's datatype
|
|
789
|
+
through the SAME renderer: written verbatim it is a name the document never
|
|
790
|
+
declared (or, for an IRI, an unbracketed one OWL 2 does not allow). A
|
|
791
|
+
language-tagged literal has no datatype to render: ``"abc"@en``.
|
|
792
|
+
"""
|
|
793
|
+
if literal.language is not None:
|
|
794
|
+
return literal.to_unicode()
|
|
795
|
+
return f'"{_escape(literal.lexical)}"^^{name(literal.datatype)}'
|
|
796
|
+
|
|
797
|
+
|
|
798
|
+
def render_datarange_fs(datarange: DataRange, name: NameRenderer = lambda n: n) -> str:
|
|
799
|
+
"""The OWL 2 Functional-Style text of ``datarange``; ``name`` renders a
|
|
800
|
+
DATATYPE name (default: verbatim). A facet name is always written as it is:
|
|
801
|
+
a facet is part of OWL 2's own vocabulary, never a name the caller owns."""
|
|
802
|
+
if isinstance(datarange, Datatype):
|
|
803
|
+
return name(datarange.name)
|
|
804
|
+
if isinstance(datarange, DatatypeRestriction):
|
|
805
|
+
body = " ".join(f"{f} {render_literal_fs(b, name)}" for f, b in datarange.facets)
|
|
806
|
+
return f"DatatypeRestriction({name(datarange.base.name)} {body})"
|
|
807
|
+
if isinstance(datarange, DataOneOf):
|
|
808
|
+
return ("DataOneOf(" + " ".join(render_literal_fs(v, name) for v in datarange.values)
|
|
809
|
+
+ ")")
|
|
810
|
+
if isinstance(datarange, DataComplementOf):
|
|
811
|
+
return f"DataComplementOf({render_datarange_fs(datarange.datarange, name)})"
|
|
812
|
+
if isinstance(datarange, DataIntersectionOf):
|
|
813
|
+
return ("DataIntersectionOf("
|
|
814
|
+
+ " ".join(render_datarange_fs(r, name) for r in datarange.ranges) + ")")
|
|
815
|
+
if isinstance(datarange, DataUnionOf):
|
|
816
|
+
return ("DataUnionOf("
|
|
817
|
+
+ " ".join(render_datarange_fs(r, name) for r in datarange.ranges) + ")")
|
|
818
|
+
raise TypeError(f"render: unsupported data range {type(datarange).__name__}")
|