unicode-logic-kit 0.31.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. unicode_logic_kit/__init__.py +385 -0
  2. unicode_logic_kit/__main__.py +520 -0
  3. unicode_logic_kit/_deadline.py +219 -0
  4. unicode_logic_kit/ace/__init__.py +126 -0
  5. unicode_logic_kit/ace/_align.py +135 -0
  6. unicode_logic_kit/ace/chem_lexicon.py +128 -0
  7. unicode_logic_kit/ace/drs_reader.py +570 -0
  8. unicode_logic_kit/ace/mapping.py +666 -0
  9. unicode_logic_kit/ace/reverse_modal.py +138 -0
  10. unicode_logic_kit/ace/runner.py +551 -0
  11. unicode_logic_kit/ace/translate.py +452 -0
  12. unicode_logic_kit/ace/verbalize.py +1070 -0
  13. unicode_logic_kit/api.py +1284 -0
  14. unicode_logic_kit/atp/__init__.py +177 -0
  15. unicode_logic_kit/atp/_ascii_names.py +113 -0
  16. unicode_logic_kit/atp/_html.py +72 -0
  17. unicode_logic_kit/atp/_substructural_input.py +228 -0
  18. unicode_logic_kit/atp/_tff_problem.py +715 -0
  19. unicode_logic_kit/atp/_tptp_problem.py +1111 -0
  20. unicode_logic_kit/atp/_writer_support.py +289 -0
  21. unicode_logic_kit/atp/clingo_backend.py +1180 -0
  22. unicode_logic_kit/atp/cvc5_backend.py +1385 -0
  23. unicode_logic_kit/atp/eprover_backend.py +732 -0
  24. unicode_logic_kit/atp/finite_domain.py +1055 -0
  25. unicode_logic_kit/atp/fitch.py +1547 -0
  26. unicode_logic_kit/atp/fitch_search.py +551 -0
  27. unicode_logic_kit/atp/hets_backend.py +339 -0
  28. unicode_logic_kit/atp/hybrid_down.py +120 -0
  29. unicode_logic_kit/atp/incremental.py +250 -0
  30. unicode_logic_kit/atp/kripke_enum.py +741 -0
  31. unicode_logic_kit/atp/lambek.py +436 -0
  32. unicode_logic_kit/atp/leo3_backend.py +332 -0
  33. unicode_logic_kit/atp/linear.py +738 -0
  34. unicode_logic_kit/atp/lj.py +705 -0
  35. unicode_logic_kit/atp/logic_backends.py +566 -0
  36. unicode_logic_kit/atp/ltl_tableau.py +1084 -0
  37. unicode_logic_kit/atp/minizinc_backend.py +1402 -0
  38. unicode_logic_kit/atp/modal_tableau.py +1382 -0
  39. unicode_logic_kit/atp/nanocop_backend.py +410 -0
  40. unicode_logic_kit/atp/portfolio.py +489 -0
  41. unicode_logic_kit/atp/protocol.py +1803 -0
  42. unicode_logic_kit/atp/prover9_entailment.py +1153 -0
  43. unicode_logic_kit/atp/resolution.py +1376 -0
  44. unicode_logic_kit/atp/resolution_check.py +1114 -0
  45. unicode_logic_kit/atp/sequent.py +1050 -0
  46. unicode_logic_kit/atp/tableau.py +921 -0
  47. unicode_logic_kit/atp/tableau_check.py +543 -0
  48. unicode_logic_kit/atp/tptp_ncl.py +811 -0
  49. unicode_logic_kit/atp/tptp_tff.py +1546 -0
  50. unicode_logic_kit/atp/tstp.py +1333 -0
  51. unicode_logic_kit/atp/tstp_check.py +1096 -0
  52. unicode_logic_kit/atp/twee_backend.py +236 -0
  53. unicode_logic_kit/atp/twee_check.py +711 -0
  54. unicode_logic_kit/atp/twee_entailment.py +953 -0
  55. unicode_logic_kit/atp/vampire_entailment.py +540 -0
  56. unicode_logic_kit/atp/z3_arith.py +470 -0
  57. unicode_logic_kit/atp/z3_equivalence.py +36 -0
  58. unicode_logic_kit/atp/z3_fuzzy.py +362 -0
  59. unicode_logic_kit/atp/z3_input.py +500 -0
  60. unicode_logic_kit/atp/z3_models.py +208 -0
  61. unicode_logic_kit/chem/__init__.py +88 -0
  62. unicode_logic_kit/chem/_naming.py +284 -0
  63. unicode_logic_kit/chem/cache.py +185 -0
  64. unicode_logic_kit/chem/interop.py +244 -0
  65. unicode_logic_kit/chem/mol.py +525 -0
  66. unicode_logic_kit/chem/signature.py +112 -0
  67. unicode_logic_kit/comorphism.py +497 -0
  68. unicode_logic_kit/dl/__init__.py +384 -0
  69. unicode_logic_kit/dl/classification.py +227 -0
  70. unicode_logic_kit/dl/concepts.py +632 -0
  71. unicode_logic_kit/dl/datatypes.py +818 -0
  72. unicode_logic_kit/dl/owl_functional.py +2433 -0
  73. unicode_logic_kit/dl/owl_manchester.py +1637 -0
  74. unicode_logic_kit/dl/owl_reasoner.py +790 -0
  75. unicode_logic_kit/dl/parser.py +391 -0
  76. unicode_logic_kit/dl/tableau.py +4048 -0
  77. unicode_logic_kit/dl/translate.py +2704 -0
  78. unicode_logic_kit/drt/__init__.py +94 -0
  79. unicode_logic_kit/drt/export.py +179 -0
  80. unicode_logic_kit/drt/nodes.py +506 -0
  81. unicode_logic_kit/drt/parser.py +965 -0
  82. unicode_logic_kit/drt/resolve.py +195 -0
  83. unicode_logic_kit/drt/reverse.py +175 -0
  84. unicode_logic_kit/eval/__init__.py +106 -0
  85. unicode_logic_kit/eval/batch.py +382 -0
  86. unicode_logic_kit/eval/canonical.py +663 -0
  87. unicode_logic_kit/eval/chem_batch.py +606 -0
  88. unicode_logic_kit/eval/converses.py +200 -0
  89. unicode_logic_kit/eval/datasets/__init__.py +136 -0
  90. unicode_logic_kit/eval/datasets/_base.py +263 -0
  91. unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
  92. unicode_logic_kit/eval/datasets/c3po.py +678 -0
  93. unicode_logic_kit/eval/datasets/folio.py +158 -0
  94. unicode_logic_kit/eval/datasets/fracas.py +418 -0
  95. unicode_logic_kit/eval/datasets/groves.py +191 -0
  96. unicode_logic_kit/eval/datasets/logicbench.py +467 -0
  97. unicode_logic_kit/eval/datasets/logicnli.py +303 -0
  98. unicode_logic_kit/eval/datasets/malls.py +133 -0
  99. unicode_logic_kit/eval/datasets/pfolio.py +594 -0
  100. unicode_logic_kit/eval/datasets/pmb.py +242 -0
  101. unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
  102. unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
  103. unicode_logic_kit/eval/datasets/proverqa.py +674 -0
  104. unicode_logic_kit/eval/datasets/willow.py +478 -0
  105. unicode_logic_kit/eval/equivalence.py +466 -0
  106. unicode_logic_kit/eval/exercise_gen.py +533 -0
  107. unicode_logic_kit/eval/explain.py +791 -0
  108. unicode_logic_kit/eval/generality.py +750 -0
  109. unicode_logic_kit/eval/metric_hf.py +458 -0
  110. unicode_logic_kit/eval/predicate_match.py +343 -0
  111. unicode_logic_kit/eval/theory_check.py +1170 -0
  112. unicode_logic_kit/eval/validate.py +306 -0
  113. unicode_logic_kit/fol/__init__.py +177 -0
  114. unicode_logic_kit/fol/_atom_keys.py +510 -0
  115. unicode_logic_kit/fol/_fol_nodes.py +3586 -0
  116. unicode_logic_kit/fol/_free_parameters.py +105 -0
  117. unicode_logic_kit/fol/_ho_nodes.py +448 -0
  118. unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
  119. unicode_logic_kit/fol/_identifiers.py +1091 -0
  120. unicode_logic_kit/fol/_lambek_nodes.py +112 -0
  121. unicode_logic_kit/fol/_linear_nodes.py +352 -0
  122. unicode_logic_kit/fol/_modal_nodes.py +1467 -0
  123. unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
  124. unicode_logic_kit/fol/_numeral_symbols.py +231 -0
  125. unicode_logic_kit/fol/_so_nodes.py +200 -0
  126. unicode_logic_kit/fol/_symbol_names.py +81 -0
  127. unicode_logic_kit/fol/_team_nodes.py +181 -0
  128. unicode_logic_kit/fol/_tptp_symbols.py +551 -0
  129. unicode_logic_kit/fol/_truth_constants.py +117 -0
  130. unicode_logic_kit/fol/casl_export.py +1135 -0
  131. unicode_logic_kit/fol/casl_import.py +929 -0
  132. unicode_logic_kit/fol/derivation.py +367 -0
  133. unicode_logic_kit/fol/dialect_detect.py +70 -0
  134. unicode_logic_kit/fol/dialect_repair.py +537 -0
  135. unicode_logic_kit/fol/frames.py +637 -0
  136. unicode_logic_kit/fol/grammars/terminals.lark +31 -0
  137. unicode_logic_kit/fol/lambda_tools.py +297 -0
  138. unicode_logic_kit/fol/latex_input.py +429 -0
  139. unicode_logic_kit/fol/modal_translation.py +944 -0
  140. unicode_logic_kit/fol/msflparser.py +1033 -0
  141. unicode_logic_kit/fol/naming.py +422 -0
  142. unicode_logic_kit/fol/nodes.py +241 -0
  143. unicode_logic_kit/fol/normalforms.py +492 -0
  144. unicode_logic_kit/fol/pal.py +287 -0
  145. unicode_logic_kit/fol/prolog_export.py +566 -0
  146. unicode_logic_kit/fol/prolog_input.py +505 -0
  147. unicode_logic_kit/fol/prover9_input.py +1325 -0
  148. unicode_logic_kit/fol/qml.py +1760 -0
  149. unicode_logic_kit/fol/qmltp_input.py +525 -0
  150. unicode_logic_kit/fol/sanitize.py +221 -0
  151. unicode_logic_kit/fol/serialize.py +79 -0
  152. unicode_logic_kit/fol/signature.py +1290 -0
  153. unicode_logic_kit/fol/simplify_check.py +544 -0
  154. unicode_logic_kit/fol/spans.py +594 -0
  155. unicode_logic_kit/fol/tptp_input.py +1503 -0
  156. unicode_logic_kit/fol/tptp_repair.py +941 -0
  157. unicode_logic_kit/fol/unification.py +157 -0
  158. unicode_logic_kit/fol/verbalize.py +263 -0
  159. unicode_logic_kit/hets/__init__.py +163 -0
  160. unicode_logic_kit/hets/bridge.py +142 -0
  161. unicode_logic_kit/hets/client.py +748 -0
  162. unicode_logic_kit/hets/docker.py +420 -0
  163. unicode_logic_kit/hets/dol.py +712 -0
  164. unicode_logic_kit/hets/haskell_json.py +355 -0
  165. unicode_logic_kit/hets/owl_backend.py +794 -0
  166. unicode_logic_kit/hets/owl_cli.py +598 -0
  167. unicode_logic_kit/hets/symbols.py +512 -0
  168. unicode_logic_kit/hol/__init__.py +140 -0
  169. unicode_logic_kit/hol/_ho_common.py +323 -0
  170. unicode_logic_kit/hol/_isabelle_binders.py +125 -0
  171. unicode_logic_kit/hol/classical.py +812 -0
  172. unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
  173. unicode_logic_kit/hol/deepshallow/_common.py +177 -0
  174. unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
  175. unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
  176. unicode_logic_kit/hol/deepshallow/modal.py +217 -0
  177. unicode_logic_kit/hol/deepshallow/qml.py +406 -0
  178. unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
  179. unicode_logic_kit/hol/free.py +753 -0
  180. unicode_logic_kit/hol/goedel.py +336 -0
  181. unicode_logic_kit/hol/ho_modal.py +1743 -0
  182. unicode_logic_kit/hol/intuitionistic.py +403 -0
  183. unicode_logic_kit/hol/isabelle_conditional.py +593 -0
  184. unicode_logic_kit/hol/isabelle_modal.py +1908 -0
  185. unicode_logic_kit/hol/isabelle_relevant.py +412 -0
  186. unicode_logic_kit/hol/isabelle_runner.py +1147 -0
  187. unicode_logic_kit/hol/isabelle_substructural.py +884 -0
  188. unicode_logic_kit/hol/lean.py +1018 -0
  189. unicode_logic_kit/hol/manyvalued.py +921 -0
  190. unicode_logic_kit/hol/secondorder.py +687 -0
  191. unicode_logic_kit/hol/thf_modal.py +941 -0
  192. unicode_logic_kit/hol/thirdorder.py +397 -0
  193. unicode_logic_kit/ilp/__init__.py +89 -0
  194. unicode_logic_kit/ilp/readback.py +389 -0
  195. unicode_logic_kit/ilp/separation.py +153 -0
  196. unicode_logic_kit/ilp/task.py +730 -0
  197. unicode_logic_kit/logic.py +163 -0
  198. unicode_logic_kit/mcp/__init__.py +28 -0
  199. unicode_logic_kit/mcp/__main__.py +5 -0
  200. unicode_logic_kit/mcp/chem_tools.py +1031 -0
  201. unicode_logic_kit/mcp/server.py +2453 -0
  202. unicode_logic_kit/mcp/syntax_spec.py +681 -0
  203. unicode_logic_kit/prob/__init__.py +53 -0
  204. unicode_logic_kit/prob/_bdd.py +225 -0
  205. unicode_logic_kit/prob/_column_gen.py +668 -0
  206. unicode_logic_kit/prob/distribution.py +686 -0
  207. unicode_logic_kit/prob/nilsson.py +470 -0
  208. unicode_logic_kit/py.typed +0 -0
  209. unicode_logic_kit/semantics/__init__.py +137 -0
  210. unicode_logic_kit/semantics/_modal_reject.py +156 -0
  211. unicode_logic_kit/semantics/action_models.py +466 -0
  212. unicode_logic_kit/semantics/asp_models.py +1200 -0
  213. unicode_logic_kit/semantics/conditional.py +580 -0
  214. unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
  215. unicode_logic_kit/semantics/free_logic.py +913 -0
  216. unicode_logic_kit/semantics/fuzzy.py +384 -0
  217. unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
  218. unicode_logic_kit/semantics/intuitionistic.py +581 -0
  219. unicode_logic_kit/semantics/kripke.py +1139 -0
  220. unicode_logic_kit/semantics/manyvalued.py +580 -0
  221. unicode_logic_kit/semantics/matrix.py +342 -0
  222. unicode_logic_kit/semantics/model_eval.py +1135 -0
  223. unicode_logic_kit/semantics/modelfinder.py +1036 -0
  224. unicode_logic_kit/semantics/nonmonotonic.py +372 -0
  225. unicode_logic_kit/semantics/relevant.py +331 -0
  226. unicode_logic_kit/semantics/secondorder.py +657 -0
  227. unicode_logic_kit/semantics/structures.py +352 -0
  228. unicode_logic_kit/semantics/tarski.py +975 -0
  229. unicode_logic_kit/semantics/team.py +315 -0
  230. unicode_logic_kit/semantics/team_translation.py +416 -0
  231. unicode_logic_kit/semantics/thirdorder.py +358 -0
  232. unicode_logic_kit/semantics/tnorm.py +85 -0
  233. unicode_logic_kit/semantics/truthtable.py +201 -0
  234. unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
  235. unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
  236. unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
  237. unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,818 @@
1
+ """Datatypes, literals and data ranges: the data half of the OWL 2 DL layer.
2
+
3
+ OWL 2 has two disjoint domains: the OBJECT domain (individuals, ``Δ_I``) and
4
+ the DATA domain (``Δ_D``: numbers, strings, dates, … — the *data values*).
5
+ Classes and object properties live over the first, datatypes and data
6
+ properties relate the first to the second. This module holds the pieces of the
7
+ second that the rest of :mod:`unicode_logic_kit.dl` needs:
8
+
9
+ * :class:`Literal` — a typed literal such as ``"400"^^xsd:integer``, and the
10
+ term of the first-order image it becomes (:meth:`Literal.to_term`);
11
+ * the DATA-RANGE AST (:class:`Datatype`, :class:`DatatypeRestriction`,
12
+ :class:`DataOneOf`, :class:`DataComplementOf`, :class:`DataIntersectionOf`,
13
+ :class:`DataUnionOf`) and its first-order image
14
+ (:func:`datarange_to_fol`);
15
+ * the tables the two-sorted image reads (:func:`datatype_ancestors`,
16
+ :func:`datatype_family`) and the two reserved guard predicates
17
+ :data:`OWL_THING` / :data:`OWL_DATA`.
18
+
19
+ The image is GUARDED ONE-SORTED first-order logic
20
+ ---------------------------------------------------
21
+ Not the kit's many-sorted logic (``SortedQuantifier``/``SortedConstant``): a
22
+ sorted constant must be lower-case, and every OEO individual and literal is
23
+ CamelCase. A datatype ``D`` becomes the unary predicate ``D(v)``; the object
24
+ domain and the data domain become the two reserved unary predicates
25
+ ``OwlThing`` and ``OwlData``, and ``rdfs:Literal`` — the datatype whose value
26
+ space IS the data domain — maps onto ``OwlData`` exactly as ``owl:Thing``
27
+ already maps onto ``⊤``. The axioms that make the two predicates a
28
+ two-sorted theory (their disjointness, the typing of every property, the
29
+ datatype lattice, …) are SIDE axioms of the knowledge base, never conjuncts of
30
+ its formula: see :func:`unicode_logic_kit.dl.translate.data_sort_axioms`.
31
+
32
+ What a literal becomes
33
+ -----------------------
34
+ OWL's literal-to-value map is fixed, so a literal denotes ONE data value and
35
+ the image needs a TERM for it that is the same term for the same value and a
36
+ different one for a different value:
37
+
38
+ * a literal of the exact-number family (``xsd:integer`` and its subtypes,
39
+ ``xsd:decimal``) becomes ``Number(value)`` — ``"1"^^xsd:integer`` and
40
+ ``"1.0"^^xsd:decimal`` are the SAME data value in OWL 2 (the value spaces
41
+ overlap), so both become ``Number(1)``. A decimal that no kit ``Number``
42
+ can hold EXACTLY (more than 15 significant digits, where two different
43
+ decimals can be one float) is refused by name rather than read as another
44
+ number;
45
+ * a literal of ``xsd:float``/``xsd:double`` is REFUSED by name: their value
46
+ spaces are disjoint from the exact numbers (``-0``, ``NaN``, ``±INF``), so
47
+ ``Number(1)`` would conflate two different values and any other term would
48
+ be a made-up one;
49
+ * ``xsd:normalizedString`` and ``xsd:token`` accept any string once XSD's
50
+ whitespace processing (§4.3.6) has run — *replace* (tab, line feed and
51
+ carriage return become a space) for the first, *collapse* (replace, then runs
52
+ of spaces become one and the ends are trimmed) for the second — so the literal
53
+ denotes the same VALUE as the ``xsd:string`` of the processed text and gets
54
+ that ``xsd:string`` term, typed by the datatype it was written with:
55
+ ``" a b "^^xsd:token`` is ``"a b"^^xsd:string``. ``xsd:anyURI`` is a family
56
+ of its own, disjoint from the strings, whose term is the collapsed text: two
57
+ different texts are two different values. ``xsd:language``, ``xsd:Name``,
58
+ ``xsd:NCName`` and ``xsd:NMTOKEN`` have lexical spaces (the XML name
59
+ productions) this kit does not validate, so it can neither say that a literal
60
+ of one denotes a value at all nor that it is the ``xsd:token`` of the same
61
+ text: they are REFUSED by name, like ``xsd:float``;
62
+ * every other literal becomes ``Constant('"abc"^^xsd:string')`` — its own
63
+ OWL text, kept verbatim, which is the convention :mod:`unicode_logic_kit.dl`
64
+ already follows for IRI-shaped names. The unicode syntax writes a constant of
65
+ any name in single quotes when its bare word would not read back as that
66
+ constant, so this one prints ``'"abc"^^xsd:string'`` and reads back through
67
+ ``api.parse_any`` as the very constant (a single quote inside the lexical form
68
+ is escaped with a backslash between the quotes). What stays a documented limit
69
+ of the printed form, and a limit of RULE 5 (what the kit prints reads back
70
+ through ``api.parse_any``) that holds BY DESIGN for the data layer, is the name
71
+ of a built-in datatype: it is a PREDICATE, a predicate has no quoted form, and
72
+ so every image that names one (``xsd:integer(x0)``) prints text
73
+ ``api.parse_any`` rejects, which ``tests/test_printed_text_reads_back.py``
74
+ carves out. The AST route (``api.prove`` over the nodes) is unaffected. To
75
+ print such an image as text the kit reads, rename its symbols with
76
+ :func:`unicode_logic_kit.fol.sanitize.sanitize_all` over the WHOLE premise list
77
+ with one shared mapping — ``sanitize_names`` applied to each formula with a
78
+ fresh mapping gives ``xsd:integer`` and a class called ``Xsdinteger`` the same
79
+ token, which is not injective.
80
+
81
+ The scope of facets
82
+ --------------------
83
+ The four ordering facets ``xsd:minInclusive`` / ``xsd:maxInclusive`` /
84
+ ``xsd:minExclusive`` / ``xsd:maxExclusive`` on an exact-number base datatype
85
+ are the only facets with an image this kit can interpret: they become the
86
+ native comparison atoms ``≥ ≤ > <``. Every other facet (``xsd:pattern``, the
87
+ length facets, ``xsd:totalDigits``/``xsd:fractionDigits``, ``rdf:langRange``)
88
+ and an ordering facet on a non-numeric base (``xsd:dateTime``) is REFUSED by
89
+ name: its image would be an uninterpreted function or predicate that
90
+ constrains nothing while looking as though it did, which is exactly the silent
91
+ approximation this kit does not do.
92
+ """
93
+
94
+ import re
95
+ from dataclasses import dataclass
96
+ from decimal import Decimal, InvalidOperation
97
+ from typing import Callable, Dict, FrozenSet, Iterable, Optional, Tuple
98
+
99
+ from ..fol._fol_nodes import _numeral_from_text
100
+ from ..fol.nodes import And, Atom, Constant, Node, Not, Number, Or
101
+
102
+ __all__ = [
103
+ "Literal", "DataRange", "Datatype", "DatatypeRestriction", "DataOneOf",
104
+ "DataComplementOf", "DataIntersectionOf", "DataUnionOf",
105
+ "UnsupportedDatatypeError", "datarange_to_fol", "OWL_THING", "OWL_DATA",
106
+ ]
107
+
108
+ #: The reserved unary predicate of the OBJECT domain (``Δ_I``) in the
109
+ #: two-sorted image, and of the DATA domain (``Δ_D``). Fixed names, in the
110
+ #: way ``fol.qml`` fixes ``World``/``Object``: a knowledge base that uses
111
+ #: either as a class, role, property, individual or datatype name is refused
112
+ #: by name when the two-sorted axioms are asked for.
113
+ OWL_THING = "OwlThing"
114
+ OWL_DATA = "OwlData"
115
+
116
+
117
+ class UnsupportedDatatypeError(ValueError):
118
+ """Raised for a datatype construct that is valid OWL 2 but has no faithful
119
+ first-order image in this kit: an out-of-scope facet, an ordering facet on
120
+ a non-numeric base, a literal whose value the kit cannot represent
121
+ exactly, an ill-typed literal, or a knowledge base that uses a name the
122
+ two-sorted image reserves.
123
+
124
+ Always a refusal BY NAME, with the construct, the reason and the spelling
125
+ to use instead — never an approximation.
126
+ """
127
+
128
+
129
+ # --------------------------------------------------------------------------- #
130
+ # Names
131
+ # --------------------------------------------------------------------------- #
132
+
133
+ _NAMESPACES = (
134
+ ("http://www.w3.org/2001/XMLSchema#", "xsd:"),
135
+ ("http://www.w3.org/2000/01/rdf-schema#", "rdfs:"),
136
+ ("http://www.w3.org/1999/02/22-rdf-syntax-ns#", "rdf:"),
137
+ ("http://www.w3.org/2002/07/owl#", "owl:"),
138
+ )
139
+
140
+
141
+ def canonical_datatype_name(name: str) -> str:
142
+ """``name`` with a built-in namespace written as its usual prefix:
143
+ ``http://www.w3.org/2001/XMLSchema#integer`` -- bracketed or not, as an IRI
144
+ is written in Manchester and Functional-Style Syntax -- and ``xsd:integer``
145
+ are one datatype, so they must be one NAME. Any other name is returned
146
+ unchanged (brackets and all: a user datatype's IRI is its own name).
147
+ """
148
+ bare = name[1:-1] if name[:1] == "<" and name[-1:] == ">" else name
149
+ for namespace, prefix in _NAMESPACES:
150
+ if bare.startswith(namespace):
151
+ return prefix + bare[len(namespace):]
152
+ return name
153
+
154
+
155
+ _RDFS_LITERAL = "rdfs:Literal"
156
+
157
+ # Direct super-datatypes, from the OWL 2 datatype map (Structural Specification
158
+ # §4): the value space of each key is a SUBSET of its parents'.
159
+ _PARENTS: Dict[str, Tuple[str, ...]] = {
160
+ "owl:real": (),
161
+ "owl:rational": ("owl:real",),
162
+ "xsd:decimal": ("owl:rational",),
163
+ "xsd:integer": ("xsd:decimal",),
164
+ "xsd:nonNegativeInteger": ("xsd:integer",),
165
+ "xsd:positiveInteger": ("xsd:nonNegativeInteger",),
166
+ "xsd:nonPositiveInteger": ("xsd:integer",),
167
+ "xsd:negativeInteger": ("xsd:nonPositiveInteger",),
168
+ "xsd:long": ("xsd:integer",),
169
+ "xsd:int": ("xsd:long",),
170
+ "xsd:short": ("xsd:int",),
171
+ "xsd:byte": ("xsd:short",),
172
+ "xsd:unsignedLong": ("xsd:nonNegativeInteger",),
173
+ "xsd:unsignedInt": ("xsd:unsignedLong",),
174
+ "xsd:unsignedShort": ("xsd:unsignedInt",),
175
+ "xsd:unsignedByte": ("xsd:unsignedShort",),
176
+ "rdf:PlainLiteral": (),
177
+ "xsd:string": ("rdf:PlainLiteral",),
178
+ "xsd:normalizedString": ("xsd:string",),
179
+ "xsd:token": ("xsd:normalizedString",),
180
+ "xsd:language": ("xsd:token",),
181
+ "xsd:Name": ("xsd:token",),
182
+ "xsd:NCName": ("xsd:Name",),
183
+ "xsd:NMTOKEN": ("xsd:token",),
184
+ "xsd:double": (),
185
+ "xsd:float": (),
186
+ "xsd:boolean": (),
187
+ "xsd:hexBinary": (),
188
+ "xsd:base64Binary": (),
189
+ "xsd:anyURI": (),
190
+ "xsd:dateTime": (),
191
+ "xsd:dateTimeStamp": ("xsd:dateTime",),
192
+ "rdf:XMLLiteral": (),
193
+ }
194
+
195
+ # The datatypes whose value spaces are pairwise DISJOINT in OWL 2: one entry
196
+ # per "family root". Two built-in datatypes of different families share no
197
+ # value (OWL 2 §4: even xsd:float/xsd:double are disjoint from owl:real and
198
+ # from each other). Within a family nothing is claimed — `xsd:Name` and
199
+ # `xsd:NMTOKEN` overlap, `xsd:positiveInteger` and `xsd:negativeInteger` do not,
200
+ # and the lattice edges are the only within-family facts the image states.
201
+ _FAMILY_ROOTS = ("owl:real", "xsd:double", "xsd:float", "rdf:PlainLiteral",
202
+ "xsd:boolean", "xsd:hexBinary", "xsd:base64Binary",
203
+ "xsd:anyURI", "xsd:dateTime", "rdf:XMLLiteral")
204
+
205
+ #: Every datatype of the OWL 2 datatype map this module knows, plus ``rdfs:Literal``.
206
+ BUILTIN_DATATYPES: FrozenSet[str] = frozenset(_PARENTS) | {_RDFS_LITERAL}
207
+
208
+
209
+ def datatype_ancestors(name: str) -> FrozenSet[str]:
210
+ """Every built-in datatype that strictly CONTAINS ``name``'s value space
211
+ (transitively, ``rdfs:Literal`` excluded — it contains everything). Empty
212
+ for a name outside the OWL 2 datatype map: a user datatype's relation to
213
+ the lattice is whatever its own ``DatatypeDefinition`` says.
214
+ """
215
+ name = canonical_datatype_name(name)
216
+ found = set()
217
+ stack = list(_PARENTS.get(name, ()))
218
+ while stack:
219
+ parent = stack.pop()
220
+ if parent not in found:
221
+ found.add(parent)
222
+ stack.extend(_PARENTS.get(parent, ()))
223
+ return frozenset(found)
224
+
225
+
226
+ def datatype_family(name: str) -> Optional[str]:
227
+ """The family root (one of :data:`_FAMILY_ROOTS`) a built-in datatype
228
+ belongs to, or ``None`` for ``rdfs:Literal`` and every name outside the
229
+ OWL 2 datatype map. Two datatypes with DIFFERENT non-``None`` families are
230
+ disjoint.
231
+ """
232
+ name = canonical_datatype_name(name)
233
+ if name not in _PARENTS:
234
+ return None
235
+ for root in _FAMILY_ROOTS:
236
+ if name == root or root in datatype_ancestors(name):
237
+ return root
238
+ return None
239
+
240
+
241
+ def is_builtin_datatype(name: str) -> bool:
242
+ """True iff ``name`` is in the OWL 2 datatype map (or is ``rdfs:Literal``)."""
243
+ return canonical_datatype_name(name) in BUILTIN_DATATYPES
244
+
245
+
246
+ # --------------------------------------------------------------------------- #
247
+ # Literals
248
+ # --------------------------------------------------------------------------- #
249
+
250
+ # (min, max) of each integer-valued datatype; ``None`` is unbounded.
251
+ _INTEGER_TYPES: Dict[str, Tuple[Optional[int], Optional[int]]] = {
252
+ "xsd:integer": (None, None),
253
+ "xsd:nonNegativeInteger": (0, None),
254
+ "xsd:positiveInteger": (1, None),
255
+ "xsd:nonPositiveInteger": (None, 0),
256
+ "xsd:negativeInteger": (None, -1),
257
+ "xsd:long": (-2 ** 63, 2 ** 63 - 1),
258
+ "xsd:int": (-2 ** 31, 2 ** 31 - 1),
259
+ "xsd:short": (-2 ** 15, 2 ** 15 - 1),
260
+ "xsd:byte": (-2 ** 7, 2 ** 7 - 1),
261
+ "xsd:unsignedLong": (0, 2 ** 64 - 1),
262
+ "xsd:unsignedInt": (0, 2 ** 32 - 1),
263
+ "xsd:unsignedShort": (0, 2 ** 16 - 1),
264
+ "xsd:unsignedByte": (0, 2 ** 8 - 1),
265
+ }
266
+
267
+ #: The datatypes whose literals are EXACT numbers: the integer family and
268
+ #: ``xsd:decimal``. A literal of one of them becomes a ``Number``.
269
+ EXACT_NUMBER_DATATYPES: FrozenSet[str] = frozenset(_INTEGER_TYPES) | {"xsd:decimal"}
270
+
271
+ #: The datatypes an ordering facet may restrict: the exact numbers, plus the two
272
+ #: umbrella datatypes ``owl:real``/``owl:rational`` (which have no literals of
273
+ #: their own but are legal bases).
274
+ NUMERIC_BASES: FrozenSet[str] = EXACT_NUMBER_DATATYPES | {"owl:real", "owl:rational"}
275
+
276
+ _INTEGER_LEXICAL = re.compile(r"[+-]?[0-9]+")
277
+ _DECIMAL_LEXICAL = re.compile(r"[+-]?(?:[0-9]+(?:\.[0-9]*)?|\.[0-9]+)")
278
+ _FLOATING = frozenset({"xsd:float", "xsd:double"})
279
+
280
+
281
+ def _whitespace_replace(text: str) -> str:
282
+ """XSD 1.1 §4.3.6 ``whiteSpace = replace``: tab, line feed and carriage
283
+ return each become a space. (Exactly those three: Python's ``str.split`` and
284
+ ``strip`` also treat form feed, NBSP and the Unicode spaces as white space,
285
+ which XSD does not.)"""
286
+ return text.replace("\t", " ").replace("\n", " ").replace("\r", " ")
287
+
288
+
289
+ def _whitespace_collapse(text: str) -> str:
290
+ """XSD 1.1 §4.3.6 ``whiteSpace = collapse``: *replace*, then runs of spaces
291
+ become one space and a leading or trailing space is removed."""
292
+ return re.sub(" +", " ", _whitespace_replace(text)).strip(" ")
293
+
294
+
295
+ #: The datatypes whose lexical form is whitespace-processed before it denotes a
296
+ #: value, with the processing. ``xsd:string`` (and ``rdf:PlainLiteral``) PRESERVE.
297
+ _WHITESPACE_PROCESSING: Dict[str, Callable[[str], str]] = {
298
+ "xsd:normalizedString": _whitespace_replace,
299
+ "xsd:token": _whitespace_collapse,
300
+ "xsd:anyURI": _whitespace_collapse,
301
+ }
302
+
303
+ #: ``xsd:string`` subtypes with no lexical constraint beyond the whitespace
304
+ #: processing: their literal denotes the ``xsd:string`` value of the processed text.
305
+ _STRING_VALUED = frozenset({"xsd:normalizedString", "xsd:token"})
306
+
307
+ #: Datatypes whose lexical space is an XML name production this kit does not
308
+ #: validate: a literal of one has no term (see :meth:`Literal.to_term`).
309
+ _UNVALIDATED_LEXICAL = frozenset({"xsd:language", "xsd:Name", "xsd:NCName",
310
+ "xsd:NMTOKEN"})
311
+ _BOOLEAN_CANONICAL = {"true": "true", "1": "true", "false": "false", "0": "false"}
312
+
313
+
314
+ def _escape(lexical: str) -> str:
315
+ """The OWL 2 Functional-Style string escape: a backslash before ``\\`` and ``"``."""
316
+ return lexical.replace("\\", "\\\\").replace('"', '\\"')
317
+
318
+
319
+ @dataclass(frozen=True)
320
+ class Literal:
321
+ """A typed literal ``"lexical"^^datatype`` (or a language-tagged string
322
+ ``"lexical"@language``), OWL's ``DataPropertyAssertion`` value and the
323
+ value operand of ``DataHasValue``/``DataOneOf``/a facet.
324
+
325
+ Args:
326
+ lexical: The lexical form, as written (``"400"``).
327
+ datatype: The datatype name; a built-in namespace is canonicalised to
328
+ its prefix, so ``<http://www.w3.org/2001/XMLSchema#integer>`` and
329
+ ``xsd:integer`` are one datatype. Defaults to ``xsd:string``.
330
+ language: A language tag, which makes the literal an ``rdf:PlainLiteral``
331
+ (the only datatype that has one); stored lower-case, as language
332
+ tags are case-insensitive.
333
+
334
+ Raises:
335
+ UnsupportedDatatypeError: the lexical form is ILL-TYPED for an
336
+ exact-number datatype or for ``xsd:boolean`` (``"abc"^^xsd:integer``
337
+ has no data value — an ontology containing it is inconsistent in
338
+ OWL 2 and the kit does not guess), or a language tag is given on a
339
+ datatype that has none. Other datatypes are not checked: only for
340
+ the exact numbers, ``xsd:boolean`` and strings does this kit
341
+ interpret the lexical form at all.
342
+ """
343
+
344
+ lexical: str
345
+ datatype: str = "xsd:string"
346
+ language: Optional[str] = None
347
+
348
+ def __post_init__(self):
349
+ datatype = canonical_datatype_name(self.datatype)
350
+ if self.language is not None:
351
+ if datatype not in ("xsd:string", "rdf:PlainLiteral"):
352
+ raise UnsupportedDatatypeError(
353
+ f"dl.Literal: a language tag ({self.language!r}) belongs to "
354
+ f"rdf:PlainLiteral, not to {datatype!r}. Write the literal "
355
+ f"as \"{self.lexical}\"@{self.language} or drop the tag.")
356
+ datatype = "rdf:PlainLiteral"
357
+ object.__setattr__(self, "language", self.language.lower())
358
+ elif datatype == "rdf:PlainLiteral":
359
+ raise UnsupportedDatatypeError(
360
+ f"dl.Literal: a rdf:PlainLiteral literal {self.lexical!r} without "
361
+ f"a language tag is the xsd:string {self.lexical!r}; write it as "
362
+ f"an xsd:string, or give a language tag.")
363
+ object.__setattr__(self, "datatype", datatype)
364
+ self._check_well_typed()
365
+
366
+ def _check_well_typed(self) -> None:
367
+ text = self.lexical.strip()
368
+ if self.datatype in _INTEGER_TYPES:
369
+ low, high = _INTEGER_TYPES[self.datatype]
370
+ if not _INTEGER_LEXICAL.fullmatch(text) or not (
371
+ (low is None or int(text) >= low)
372
+ and (high is None or int(text) <= high)):
373
+ raise UnsupportedDatatypeError(
374
+ f"dl.Literal: {self.lexical!r} is not a well-typed "
375
+ f"{self.datatype} literal (an ill-typed literal denotes no "
376
+ f"data value, and an ontology containing one is inconsistent "
377
+ f"in OWL 2 — this kit refuses it rather than guess).")
378
+ elif self.datatype == "xsd:decimal":
379
+ if not _DECIMAL_LEXICAL.fullmatch(text):
380
+ raise UnsupportedDatatypeError(
381
+ f"dl.Literal: {self.lexical!r} is not a well-typed "
382
+ f"xsd:decimal literal (digits with an optional sign and "
383
+ f"decimal point; no exponent).")
384
+ elif self.datatype == "xsd:boolean":
385
+ if text not in _BOOLEAN_CANONICAL:
386
+ raise UnsupportedDatatypeError(
387
+ f"dl.Literal: {self.lexical!r} is not a well-typed "
388
+ f"xsd:boolean literal (one of true, false, 1, 0).")
389
+
390
+ # -- text ------------------------------------------------------------- #
391
+
392
+ def canonical_lexical(self) -> str:
393
+ """The lexical form the TERM is named after: ``xsd:boolean`` is folded
394
+ onto ``true``/``false`` (``"1"`` and ``"true"`` are one value), the three
395
+ datatypes whose whitespace facet is not *preserve* are folded onto the
396
+ text XSD's whitespace processing makes of the lexical form
397
+ (``xsd:normalizedString`` *replace*, ``xsd:token`` and ``xsd:anyURI``
398
+ *collapse*) and every other datatype keeps the lexical form it was
399
+ written with.
400
+ """
401
+ if self.datatype == "xsd:boolean":
402
+ return _BOOLEAN_CANONICAL[self.lexical.strip()]
403
+ process = _WHITESPACE_PROCESSING.get(self.datatype)
404
+ if process is not None:
405
+ return process(self.lexical)
406
+ return self.lexical
407
+
408
+ def to_unicode(self) -> str:
409
+ """The OWL 2 text of the literal: ``"400"^^xsd:integer`` or
410
+ ``"abc"@en``. A name written back in this spelling reads back as this
411
+ literal through the OWL parsers.
412
+ """
413
+ if self.language is not None:
414
+ return f'"{_escape(self.lexical)}"@{self.language}'
415
+ return f'"{_escape(self.lexical)}"^^{self.datatype}'
416
+
417
+ def __str__(self) -> str:
418
+ return self.to_unicode()
419
+
420
+ # -- value / term ----------------------------------------------------- #
421
+
422
+ def exact_number(self) -> Optional[Decimal]:
423
+ """The literal's value as an exact :class:`~decimal.Decimal` if its
424
+ datatype is in the exact-number family, else ``None``."""
425
+ if self.datatype not in EXACT_NUMBER_DATATYPES:
426
+ return None
427
+ try:
428
+ return Decimal(self.lexical.strip())
429
+ except InvalidOperation: # pragma: no cover - checked at construction
430
+ return None
431
+
432
+ def to_term(self) -> Node:
433
+ """The first-order TERM of the literal's data value.
434
+
435
+ An exact number becomes ``Number`` (an ``int`` when integral, else a
436
+ ``float`` — and only when the decimal has at most 15 significant digits,
437
+ the rule of every numeral reader of the kit: see
438
+ :func:`~unicode_logic_kit.fol._fol_nodes._numeral_from_text`).
439
+ ``xsd:float``/``xsd:double``, a decimal no ``Number`` can
440
+ hold exactly and the datatypes whose lexical space this kit does not
441
+ validate (``xsd:language``, ``xsd:Name``, ``xsd:NCName``,
442
+ ``xsd:NMTOKEN``) are REFUSED. ``xsd:normalizedString`` and ``xsd:token``
443
+ become the ``xsd:string`` term of their whitespace-processed text.
444
+ Everything else becomes a ``Constant`` named by the literal's own OWL
445
+ text (see the module docstring).
446
+
447
+ Raises:
448
+ UnsupportedDatatypeError: see above.
449
+ """
450
+ if self.datatype in _FLOATING:
451
+ raise UnsupportedDatatypeError(
452
+ f"dl.Literal.to_term: the literal {self.to_unicode()} has no "
453
+ f"first-order image in this kit. The value space of "
454
+ f"{self.datatype} is disjoint from the exact numbers in OWL 2 "
455
+ f"(it has -0, NaN and ±INF, and 1.0 is not the integer 1), so "
456
+ f"the term Number(1.0) would conflate two different values and "
457
+ f"any other term would be invented. Use xsd:decimal or "
458
+ f"xsd:integer, or state the constraint outside the datatype.")
459
+ if self.datatype in _UNVALIDATED_LEXICAL:
460
+ raise UnsupportedDatatypeError(
461
+ f"dl.Literal.to_term: the literal {self.to_unicode()} has no "
462
+ f"first-order image in this kit. The lexical space of "
463
+ f"{self.datatype} is an XML name production this kit does not "
464
+ f"validate, so it cannot say whether the literal denotes a "
465
+ f"value at all, nor that it is the xsd:token / xsd:string value "
466
+ f"of the same text, and a term of its own would state a "
467
+ f"distinctness or an identity the value space does not "
468
+ f"guarantee. Write the value as an xsd:token or xsd:string "
469
+ f"literal, or leave it out.")
470
+ if self.datatype in _STRING_VALUED:
471
+ # The same VALUE as the xsd:string of the processed text, so the
472
+ # same term: the typing by the written datatype is a separate fact.
473
+ return Constant(Literal(self.canonical_lexical(), "xsd:string")
474
+ .to_unicode())
475
+ value = self.exact_number()
476
+ if value is None:
477
+ return Constant(Literal(self.canonical_lexical(), self.datatype,
478
+ self.language).to_unicode())
479
+ if value == value.to_integral_value():
480
+ return Number(int(value))
481
+ try:
482
+ return Number(_numeral_from_text(format(value, "f")))
483
+ except ValueError as exc:
484
+ raise UnsupportedDatatypeError(
485
+ f"dl.Literal.to_term: the decimal {self.to_unicode()} cannot be "
486
+ f"held exactly by a kit Number, and reading it as the nearest "
487
+ f"float would let two different data values collapse into one "
488
+ f"term: {exc}.") from None
489
+
490
+
491
+ # --------------------------------------------------------------------------- #
492
+ # Data ranges
493
+ # --------------------------------------------------------------------------- #
494
+
495
+ class DataRange:
496
+ """Base class for OWL 2 data ranges (see the module docstring)."""
497
+
498
+ def to_unicode(self) -> str:
499
+ """The data range in the kit's display notation (``xsd:integer[≥ 5]``,
500
+ ``{1, 2}``, ``¬D``, ``D ⊓ E``, ``D ⊔ E``)."""
501
+ return _render_range(self)
502
+
503
+ def __str__(self) -> str:
504
+ return self.to_unicode()
505
+
506
+
507
+ @dataclass(frozen=True)
508
+ class Datatype(DataRange):
509
+ """A named datatype (``xsd:integer``, ``rdfs:Literal``, a user datatype
510
+ defined by ``DatatypeDefinition``). A built-in namespace is canonicalised
511
+ to its prefix."""
512
+
513
+ name: str
514
+
515
+ def __post_init__(self):
516
+ object.__setattr__(self, "name", canonical_datatype_name(self.name))
517
+
518
+
519
+ #: The ordering facets, in canonical spelling, with the comparison atom each
520
+ #: becomes: ``value ≥ bound`` for ``xsd:minInclusive`` and so on.
521
+ _ORDER_FACETS = {
522
+ "xsd:minInclusive": "≥",
523
+ "xsd:maxInclusive": "≤",
524
+ "xsd:minExclusive": ">",
525
+ "xsd:maxExclusive": "<",
526
+ }
527
+
528
+ #: The facets this kit translates (see the module docstring's "The scope of facets").
529
+ SUPPORTED_FACETS: Tuple[str, ...] = tuple(_ORDER_FACETS)
530
+
531
+ _NO_IMAGE_FACETS = {
532
+ "xsd:pattern": "regular-expression constraints over a literal's lexical "
533
+ "space are not first-order, and no backend here interprets them",
534
+ "xsd:length": "string length is an uninterpreted function on every backend "
535
+ "here, so the axiom would constrain nothing while looking as "
536
+ "though it did",
537
+ "xsd:minLength": "string length is an uninterpreted function on every "
538
+ "backend here, so the axiom would constrain nothing while "
539
+ "looking as though it did",
540
+ "xsd:maxLength": "string length is an uninterpreted function on every "
541
+ "backend here, so the axiom would constrain nothing while "
542
+ "looking as though it did",
543
+ "xsd:totalDigits": "digit counts are not first-order over the kit's numbers",
544
+ "xsd:fractionDigits": "digit counts are not first-order over the kit's numbers",
545
+ "rdf:langRange": "language-range matching is not first-order over the "
546
+ "kit's strings",
547
+ }
548
+
549
+
550
+ def _check_facet(base: str, facet: str, bound: "Literal") -> str:
551
+ """Validate one facet of a restriction of ``base``; return its canonical name."""
552
+ facet = canonical_datatype_name(facet)
553
+ if facet not in _ORDER_FACETS:
554
+ why = _NO_IMAGE_FACETS.get(facet, "it is not one of the four ordering facets")
555
+ raise UnsupportedDatatypeError(
556
+ f"dl.datatypes: the facet {facet!r} has no first-order image in this "
557
+ f"kit — {why}. Supported facets are {', '.join(SUPPORTED_FACETS)} "
558
+ f"on an exact-number base datatype; drop the facet, or state the "
559
+ f"constraint outside the datatype.")
560
+ if base not in NUMERIC_BASES:
561
+ raise UnsupportedDatatypeError(
562
+ f"dl.datatypes: the facet {facet!r} is supported only on an "
563
+ f"exact-number base datatype (xsd:integer and its subtypes, "
564
+ f"xsd:decimal), not on {base!r} — this kit's ≤/≥ atoms carry no "
565
+ f"theory for it, so the image would be an uninterpreted predicate "
566
+ f"over uninterpreted constants and would constrain nothing. Use "
567
+ f"xsd:decimal or xsd:integer, or state the bound outside the "
568
+ f"datatype.")
569
+ if bound.exact_number() is None:
570
+ raise UnsupportedDatatypeError(
571
+ f"dl.datatypes: the bound {bound.to_unicode()} of the facet "
572
+ f"{facet!r} is not an exact number (xsd:integer family or "
573
+ f"xsd:decimal literal), so it has no ordering term in this kit.")
574
+ return facet
575
+
576
+
577
+ @dataclass(frozen=True)
578
+ class DatatypeRestriction(DataRange):
579
+ """``DatatypeRestriction(base facet₁ literal₁ …)``: the values of ``base``
580
+ that satisfy every facet. ``facets`` is a tuple of ``(facet name, bound
581
+ literal)``; it is validated against the supported scope on construction.
582
+
583
+ Raises:
584
+ UnsupportedDatatypeError: an out-of-scope facet, or an ordering facet
585
+ on a non-numeric base (see the module docstring).
586
+ """
587
+
588
+ base: Datatype
589
+ facets: Tuple[Tuple[str, Literal], ...]
590
+
591
+ def __post_init__(self):
592
+ base = self.base if isinstance(self.base, Datatype) else Datatype(self.base)
593
+ object.__setattr__(self, "base", base)
594
+ facets = tuple((facet, bound) for facet, bound in self.facets)
595
+ if not facets:
596
+ raise UnsupportedDatatypeError(
597
+ "dl.DatatypeRestriction: a restriction needs at least one facet "
598
+ "(OWL 2's grammar is DatatypeRestriction(DT (F lt)+)); a datatype "
599
+ "with no facet is just the datatype.")
600
+ checked = tuple((_check_facet(base.name, facet, bound), bound)
601
+ for facet, bound in facets)
602
+ object.__setattr__(self, "facets", checked)
603
+
604
+
605
+ @dataclass(frozen=True)
606
+ class DataOneOf(DataRange):
607
+ """``DataOneOf(lt₁ … ltₙ)``: exactly the listed data values."""
608
+
609
+ values: Tuple[Literal, ...]
610
+
611
+ def __post_init__(self):
612
+ object.__setattr__(self, "values", tuple(self.values))
613
+ if not self.values:
614
+ raise UnsupportedDatatypeError(
615
+ "dl.DataOneOf: needs at least one literal (OWL 2's grammar is "
616
+ "DataOneOf(lt+)).")
617
+
618
+
619
+ @dataclass(frozen=True)
620
+ class DataComplementOf(DataRange):
621
+ """``DataComplementOf(DR)``: the data values NOT in ``DR`` — the
622
+ complement within the data domain, not within the whole universe."""
623
+
624
+ datarange: DataRange
625
+
626
+
627
+ @dataclass(frozen=True)
628
+ class DataIntersectionOf(DataRange):
629
+ """``DataIntersectionOf(DR₁ … DRₙ)``, ``n ≥ 2``."""
630
+
631
+ ranges: Tuple[DataRange, ...]
632
+
633
+ def __post_init__(self):
634
+ object.__setattr__(self, "ranges", tuple(self.ranges))
635
+ if len(self.ranges) < 2:
636
+ raise UnsupportedDatatypeError(
637
+ "dl.DataIntersectionOf: needs at least two data ranges.")
638
+
639
+
640
+ @dataclass(frozen=True)
641
+ class DataUnionOf(DataRange):
642
+ """``DataUnionOf(DR₁ … DRₙ)``, ``n ≥ 2``."""
643
+
644
+ ranges: Tuple[DataRange, ...]
645
+
646
+ def __post_init__(self):
647
+ object.__setattr__(self, "ranges", tuple(self.ranges))
648
+ if len(self.ranges) < 2:
649
+ raise UnsupportedDatatypeError(
650
+ "dl.DataUnionOf: needs at least two data ranges.")
651
+
652
+
653
+ # --------------------------------------------------------------------------- #
654
+ # Display
655
+ # --------------------------------------------------------------------------- #
656
+
657
+ _FACET_SYMBOL = {facet: symbol for facet, symbol in _ORDER_FACETS.items()}
658
+
659
+
660
+ def _render_range(dr: DataRange) -> str:
661
+ if isinstance(dr, Datatype):
662
+ return dr.name
663
+ if isinstance(dr, DatatypeRestriction):
664
+ facets = ", ".join(f"{_FACET_SYMBOL[f]} {b.canonical_lexical()}"
665
+ for f, b in dr.facets)
666
+ return f"{dr.base.name}[{facets}]"
667
+ if isinstance(dr, DataOneOf):
668
+ return "{" + ", ".join(v.to_unicode() for v in dr.values) + "}"
669
+ if isinstance(dr, DataComplementOf):
670
+ return "¬" + _paren_range(dr.datarange)
671
+ if isinstance(dr, DataIntersectionOf):
672
+ return " ⊓ ".join(_paren_range(r) for r in dr.ranges)
673
+ if isinstance(dr, DataUnionOf):
674
+ return " ⊔ ".join(_paren_range(r) for r in dr.ranges)
675
+ raise TypeError(f"render: unsupported data range {type(dr).__name__}")
676
+
677
+
678
+ def _paren_range(dr: DataRange) -> str:
679
+ inner = _render_range(dr)
680
+ if isinstance(dr, (DataIntersectionOf, DataUnionOf)):
681
+ return f"({inner})"
682
+ return inner
683
+
684
+
685
+ # --------------------------------------------------------------------------- #
686
+ # The first-order image of a data range
687
+ # --------------------------------------------------------------------------- #
688
+
689
+ def _fold(ctor, parts: Iterable[Node]) -> Node:
690
+ parts = list(parts)
691
+ acc = parts[0]
692
+ for part in parts[1:]:
693
+ acc = ctor(acc, part)
694
+ return acc
695
+
696
+
697
+ def datarange_to_fol(datarange: DataRange, term: Node) -> Node:
698
+ """The first-order image ``δ(DR, t)`` of a data range, with ``term`` (a
699
+ ``Variable``, or any other term) standing for the data value under
700
+ discussion. OWL 2 direct semantics (Structural Specification §7):
701
+
702
+ * ``Datatype(D)`` ↦ ``D(t)``; ``rdfs:Literal`` — whose value space is the
703
+ whole data domain — ↦ ``OwlData(t)``;
704
+ * ``DatatypeRestriction(D f₁ v₁ … fₙ vₙ)`` ↦ ``D(t) ∧ t ⋈₁ v₁ ∧ … ∧ t ⋈ₙ vₙ``
705
+ (``(D)^DT ∩ ⋂ (fᵢ, vᵢ)^F`` is the conjunction of the base guard with
706
+ one comparison atom per facet);
707
+ * ``DataOneOf(l₁ … lₙ)`` ↦ ``t = l₁ ∨ … ∨ t = lₙ``;
708
+ * ``DataComplementOf(DR)`` ↦ ``OwlData(t) ∧ ¬δ(DR, t)`` — the complement
709
+ is taken WITHIN the data domain (``Δ_D \\ DR^DT``), so the ``OwlData``
710
+ conjunct is load-bearing: without it the image would also admit
711
+ individuals;
712
+ * ``DataIntersectionOf`` ↦ ``∧``, ``DataUnionOf`` ↦ ``∨``.
713
+
714
+ Raises:
715
+ UnsupportedDatatypeError: a literal with no term (``xsd:double``, an
716
+ inexact decimal).
717
+ """
718
+ if isinstance(datarange, Datatype):
719
+ if datarange.name == _RDFS_LITERAL:
720
+ return Atom(OWL_DATA, (term,))
721
+ return Atom(datarange.name, (term,))
722
+ if isinstance(datarange, DatatypeRestriction):
723
+ parts = [datarange_to_fol(datarange.base, term)]
724
+ for facet, bound in datarange.facets:
725
+ parts.append(Atom(_ORDER_FACETS[facet], (term, bound.to_term())))
726
+ return _fold(And, parts)
727
+ if isinstance(datarange, DataOneOf):
728
+ return _fold(Or, [Atom("=", (term, value.to_term()))
729
+ for value in datarange.values])
730
+ if isinstance(datarange, DataComplementOf):
731
+ return And(Atom(OWL_DATA, (term,)),
732
+ Not(datarange_to_fol(datarange.datarange, term)))
733
+ if isinstance(datarange, DataIntersectionOf):
734
+ return _fold(And, [datarange_to_fol(r, term) for r in datarange.ranges])
735
+ if isinstance(datarange, DataUnionOf):
736
+ return _fold(Or, [datarange_to_fol(r, term) for r in datarange.ranges])
737
+ raise TypeError(f"datarange_to_fol: unsupported data range {type(datarange).__name__}")
738
+
739
+
740
+ # --------------------------------------------------------------------------- #
741
+ # Walking a data range
742
+ # --------------------------------------------------------------------------- #
743
+
744
+ def datarange_literals(datarange: DataRange) -> Tuple[Literal, ...]:
745
+ """Every :class:`Literal` occurring in ``datarange`` (facet bounds and
746
+ one-of values), in order of occurrence."""
747
+ if isinstance(datarange, Datatype):
748
+ return ()
749
+ if isinstance(datarange, DatatypeRestriction):
750
+ return tuple(bound for _, bound in datarange.facets)
751
+ if isinstance(datarange, DataOneOf):
752
+ return datarange.values
753
+ if isinstance(datarange, DataComplementOf):
754
+ return datarange_literals(datarange.datarange)
755
+ if isinstance(datarange, (DataIntersectionOf, DataUnionOf)):
756
+ return tuple(lit for r in datarange.ranges for lit in datarange_literals(r))
757
+ raise TypeError(f"datarange_literals: unsupported data range {type(datarange).__name__}")
758
+
759
+
760
+ def datarange_datatypes(datarange: DataRange) -> Tuple[str, ...]:
761
+ """Every datatype NAME occurring in ``datarange`` (a base, or a plain
762
+ datatype), in order of occurrence — not the literals' datatypes, see
763
+ :func:`datarange_literals`."""
764
+ if isinstance(datarange, Datatype):
765
+ return (datarange.name,)
766
+ if isinstance(datarange, DatatypeRestriction):
767
+ return (datarange.base.name,)
768
+ if isinstance(datarange, DataOneOf):
769
+ return ()
770
+ if isinstance(datarange, DataComplementOf):
771
+ return datarange_datatypes(datarange.datarange)
772
+ if isinstance(datarange, (DataIntersectionOf, DataUnionOf)):
773
+ return tuple(name for r in datarange.ranges for name in datarange_datatypes(r))
774
+ raise TypeError(f"datarange_datatypes: unsupported data range {type(datarange).__name__}")
775
+
776
+
777
+ #: ``(callable)`` alias used by the OWL writers: maps a datatype NAME to the
778
+ #: text it is written as (``<IRI>`` brackets, a synthetic name, …).
779
+ NameRenderer = Callable[[str], str]
780
+
781
+
782
+ def render_literal_fs(literal: Literal, name: NameRenderer = lambda n: n) -> str:
783
+ """The OWL 2 Functional-Style text of ``literal`` (``"400"^^xsd:integer``);
784
+ ``name`` renders its DATATYPE name, exactly as it does the datatype names of
785
+ :func:`render_datarange_fs` (default: verbatim).
786
+
787
+ A writer that renders datatype names through a renderer of its own (an IRI in
788
+ angle brackets, a synthetic ``:T1`` token) must render a literal's datatype
789
+ through the SAME renderer: written verbatim it is a name the document never
790
+ declared (or, for an IRI, an unbracketed one OWL 2 does not allow). A
791
+ language-tagged literal has no datatype to render: ``"abc"@en``.
792
+ """
793
+ if literal.language is not None:
794
+ return literal.to_unicode()
795
+ return f'"{_escape(literal.lexical)}"^^{name(literal.datatype)}'
796
+
797
+
798
+ def render_datarange_fs(datarange: DataRange, name: NameRenderer = lambda n: n) -> str:
799
+ """The OWL 2 Functional-Style text of ``datarange``; ``name`` renders a
800
+ DATATYPE name (default: verbatim). A facet name is always written as it is:
801
+ a facet is part of OWL 2's own vocabulary, never a name the caller owns."""
802
+ if isinstance(datarange, Datatype):
803
+ return name(datarange.name)
804
+ if isinstance(datarange, DatatypeRestriction):
805
+ body = " ".join(f"{f} {render_literal_fs(b, name)}" for f, b in datarange.facets)
806
+ return f"DatatypeRestriction({name(datarange.base.name)} {body})"
807
+ if isinstance(datarange, DataOneOf):
808
+ return ("DataOneOf(" + " ".join(render_literal_fs(v, name) for v in datarange.values)
809
+ + ")")
810
+ if isinstance(datarange, DataComplementOf):
811
+ return f"DataComplementOf({render_datarange_fs(datarange.datarange, name)})"
812
+ if isinstance(datarange, DataIntersectionOf):
813
+ return ("DataIntersectionOf("
814
+ + " ".join(render_datarange_fs(r, name) for r in datarange.ranges) + ")")
815
+ if isinstance(datarange, DataUnionOf):
816
+ return ("DataUnionOf("
817
+ + " ".join(render_datarange_fs(r, name) for r in datarange.ranges) + ")")
818
+ raise TypeError(f"render: unsupported data range {type(datarange).__name__}")