unicode-logic-kit 0.31.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. unicode_logic_kit/__init__.py +385 -0
  2. unicode_logic_kit/__main__.py +520 -0
  3. unicode_logic_kit/_deadline.py +219 -0
  4. unicode_logic_kit/ace/__init__.py +126 -0
  5. unicode_logic_kit/ace/_align.py +135 -0
  6. unicode_logic_kit/ace/chem_lexicon.py +128 -0
  7. unicode_logic_kit/ace/drs_reader.py +570 -0
  8. unicode_logic_kit/ace/mapping.py +666 -0
  9. unicode_logic_kit/ace/reverse_modal.py +138 -0
  10. unicode_logic_kit/ace/runner.py +551 -0
  11. unicode_logic_kit/ace/translate.py +452 -0
  12. unicode_logic_kit/ace/verbalize.py +1070 -0
  13. unicode_logic_kit/api.py +1284 -0
  14. unicode_logic_kit/atp/__init__.py +177 -0
  15. unicode_logic_kit/atp/_ascii_names.py +113 -0
  16. unicode_logic_kit/atp/_html.py +72 -0
  17. unicode_logic_kit/atp/_substructural_input.py +228 -0
  18. unicode_logic_kit/atp/_tff_problem.py +715 -0
  19. unicode_logic_kit/atp/_tptp_problem.py +1111 -0
  20. unicode_logic_kit/atp/_writer_support.py +289 -0
  21. unicode_logic_kit/atp/clingo_backend.py +1180 -0
  22. unicode_logic_kit/atp/cvc5_backend.py +1385 -0
  23. unicode_logic_kit/atp/eprover_backend.py +732 -0
  24. unicode_logic_kit/atp/finite_domain.py +1055 -0
  25. unicode_logic_kit/atp/fitch.py +1547 -0
  26. unicode_logic_kit/atp/fitch_search.py +551 -0
  27. unicode_logic_kit/atp/hets_backend.py +339 -0
  28. unicode_logic_kit/atp/hybrid_down.py +120 -0
  29. unicode_logic_kit/atp/incremental.py +250 -0
  30. unicode_logic_kit/atp/kripke_enum.py +741 -0
  31. unicode_logic_kit/atp/lambek.py +436 -0
  32. unicode_logic_kit/atp/leo3_backend.py +332 -0
  33. unicode_logic_kit/atp/linear.py +738 -0
  34. unicode_logic_kit/atp/lj.py +705 -0
  35. unicode_logic_kit/atp/logic_backends.py +566 -0
  36. unicode_logic_kit/atp/ltl_tableau.py +1084 -0
  37. unicode_logic_kit/atp/minizinc_backend.py +1402 -0
  38. unicode_logic_kit/atp/modal_tableau.py +1382 -0
  39. unicode_logic_kit/atp/nanocop_backend.py +410 -0
  40. unicode_logic_kit/atp/portfolio.py +489 -0
  41. unicode_logic_kit/atp/protocol.py +1803 -0
  42. unicode_logic_kit/atp/prover9_entailment.py +1153 -0
  43. unicode_logic_kit/atp/resolution.py +1376 -0
  44. unicode_logic_kit/atp/resolution_check.py +1114 -0
  45. unicode_logic_kit/atp/sequent.py +1050 -0
  46. unicode_logic_kit/atp/tableau.py +921 -0
  47. unicode_logic_kit/atp/tableau_check.py +543 -0
  48. unicode_logic_kit/atp/tptp_ncl.py +811 -0
  49. unicode_logic_kit/atp/tptp_tff.py +1546 -0
  50. unicode_logic_kit/atp/tstp.py +1333 -0
  51. unicode_logic_kit/atp/tstp_check.py +1096 -0
  52. unicode_logic_kit/atp/twee_backend.py +236 -0
  53. unicode_logic_kit/atp/twee_check.py +711 -0
  54. unicode_logic_kit/atp/twee_entailment.py +953 -0
  55. unicode_logic_kit/atp/vampire_entailment.py +540 -0
  56. unicode_logic_kit/atp/z3_arith.py +470 -0
  57. unicode_logic_kit/atp/z3_equivalence.py +36 -0
  58. unicode_logic_kit/atp/z3_fuzzy.py +362 -0
  59. unicode_logic_kit/atp/z3_input.py +500 -0
  60. unicode_logic_kit/atp/z3_models.py +208 -0
  61. unicode_logic_kit/chem/__init__.py +88 -0
  62. unicode_logic_kit/chem/_naming.py +284 -0
  63. unicode_logic_kit/chem/cache.py +185 -0
  64. unicode_logic_kit/chem/interop.py +244 -0
  65. unicode_logic_kit/chem/mol.py +525 -0
  66. unicode_logic_kit/chem/signature.py +112 -0
  67. unicode_logic_kit/comorphism.py +497 -0
  68. unicode_logic_kit/dl/__init__.py +384 -0
  69. unicode_logic_kit/dl/classification.py +227 -0
  70. unicode_logic_kit/dl/concepts.py +632 -0
  71. unicode_logic_kit/dl/datatypes.py +818 -0
  72. unicode_logic_kit/dl/owl_functional.py +2433 -0
  73. unicode_logic_kit/dl/owl_manchester.py +1637 -0
  74. unicode_logic_kit/dl/owl_reasoner.py +790 -0
  75. unicode_logic_kit/dl/parser.py +391 -0
  76. unicode_logic_kit/dl/tableau.py +4048 -0
  77. unicode_logic_kit/dl/translate.py +2704 -0
  78. unicode_logic_kit/drt/__init__.py +94 -0
  79. unicode_logic_kit/drt/export.py +179 -0
  80. unicode_logic_kit/drt/nodes.py +506 -0
  81. unicode_logic_kit/drt/parser.py +965 -0
  82. unicode_logic_kit/drt/resolve.py +195 -0
  83. unicode_logic_kit/drt/reverse.py +175 -0
  84. unicode_logic_kit/eval/__init__.py +106 -0
  85. unicode_logic_kit/eval/batch.py +382 -0
  86. unicode_logic_kit/eval/canonical.py +663 -0
  87. unicode_logic_kit/eval/chem_batch.py +606 -0
  88. unicode_logic_kit/eval/converses.py +200 -0
  89. unicode_logic_kit/eval/datasets/__init__.py +136 -0
  90. unicode_logic_kit/eval/datasets/_base.py +263 -0
  91. unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
  92. unicode_logic_kit/eval/datasets/c3po.py +678 -0
  93. unicode_logic_kit/eval/datasets/folio.py +158 -0
  94. unicode_logic_kit/eval/datasets/fracas.py +418 -0
  95. unicode_logic_kit/eval/datasets/groves.py +191 -0
  96. unicode_logic_kit/eval/datasets/logicbench.py +467 -0
  97. unicode_logic_kit/eval/datasets/logicnli.py +303 -0
  98. unicode_logic_kit/eval/datasets/malls.py +133 -0
  99. unicode_logic_kit/eval/datasets/pfolio.py +594 -0
  100. unicode_logic_kit/eval/datasets/pmb.py +242 -0
  101. unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
  102. unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
  103. unicode_logic_kit/eval/datasets/proverqa.py +674 -0
  104. unicode_logic_kit/eval/datasets/willow.py +478 -0
  105. unicode_logic_kit/eval/equivalence.py +466 -0
  106. unicode_logic_kit/eval/exercise_gen.py +533 -0
  107. unicode_logic_kit/eval/explain.py +791 -0
  108. unicode_logic_kit/eval/generality.py +750 -0
  109. unicode_logic_kit/eval/metric_hf.py +458 -0
  110. unicode_logic_kit/eval/predicate_match.py +343 -0
  111. unicode_logic_kit/eval/theory_check.py +1170 -0
  112. unicode_logic_kit/eval/validate.py +306 -0
  113. unicode_logic_kit/fol/__init__.py +177 -0
  114. unicode_logic_kit/fol/_atom_keys.py +510 -0
  115. unicode_logic_kit/fol/_fol_nodes.py +3586 -0
  116. unicode_logic_kit/fol/_free_parameters.py +105 -0
  117. unicode_logic_kit/fol/_ho_nodes.py +448 -0
  118. unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
  119. unicode_logic_kit/fol/_identifiers.py +1091 -0
  120. unicode_logic_kit/fol/_lambek_nodes.py +112 -0
  121. unicode_logic_kit/fol/_linear_nodes.py +352 -0
  122. unicode_logic_kit/fol/_modal_nodes.py +1467 -0
  123. unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
  124. unicode_logic_kit/fol/_numeral_symbols.py +231 -0
  125. unicode_logic_kit/fol/_so_nodes.py +200 -0
  126. unicode_logic_kit/fol/_symbol_names.py +81 -0
  127. unicode_logic_kit/fol/_team_nodes.py +181 -0
  128. unicode_logic_kit/fol/_tptp_symbols.py +551 -0
  129. unicode_logic_kit/fol/_truth_constants.py +117 -0
  130. unicode_logic_kit/fol/casl_export.py +1135 -0
  131. unicode_logic_kit/fol/casl_import.py +929 -0
  132. unicode_logic_kit/fol/derivation.py +367 -0
  133. unicode_logic_kit/fol/dialect_detect.py +70 -0
  134. unicode_logic_kit/fol/dialect_repair.py +537 -0
  135. unicode_logic_kit/fol/frames.py +637 -0
  136. unicode_logic_kit/fol/grammars/terminals.lark +31 -0
  137. unicode_logic_kit/fol/lambda_tools.py +297 -0
  138. unicode_logic_kit/fol/latex_input.py +429 -0
  139. unicode_logic_kit/fol/modal_translation.py +944 -0
  140. unicode_logic_kit/fol/msflparser.py +1033 -0
  141. unicode_logic_kit/fol/naming.py +422 -0
  142. unicode_logic_kit/fol/nodes.py +241 -0
  143. unicode_logic_kit/fol/normalforms.py +492 -0
  144. unicode_logic_kit/fol/pal.py +287 -0
  145. unicode_logic_kit/fol/prolog_export.py +566 -0
  146. unicode_logic_kit/fol/prolog_input.py +505 -0
  147. unicode_logic_kit/fol/prover9_input.py +1325 -0
  148. unicode_logic_kit/fol/qml.py +1760 -0
  149. unicode_logic_kit/fol/qmltp_input.py +525 -0
  150. unicode_logic_kit/fol/sanitize.py +221 -0
  151. unicode_logic_kit/fol/serialize.py +79 -0
  152. unicode_logic_kit/fol/signature.py +1290 -0
  153. unicode_logic_kit/fol/simplify_check.py +544 -0
  154. unicode_logic_kit/fol/spans.py +594 -0
  155. unicode_logic_kit/fol/tptp_input.py +1503 -0
  156. unicode_logic_kit/fol/tptp_repair.py +941 -0
  157. unicode_logic_kit/fol/unification.py +157 -0
  158. unicode_logic_kit/fol/verbalize.py +263 -0
  159. unicode_logic_kit/hets/__init__.py +163 -0
  160. unicode_logic_kit/hets/bridge.py +142 -0
  161. unicode_logic_kit/hets/client.py +748 -0
  162. unicode_logic_kit/hets/docker.py +420 -0
  163. unicode_logic_kit/hets/dol.py +712 -0
  164. unicode_logic_kit/hets/haskell_json.py +355 -0
  165. unicode_logic_kit/hets/owl_backend.py +794 -0
  166. unicode_logic_kit/hets/owl_cli.py +598 -0
  167. unicode_logic_kit/hets/symbols.py +512 -0
  168. unicode_logic_kit/hol/__init__.py +140 -0
  169. unicode_logic_kit/hol/_ho_common.py +323 -0
  170. unicode_logic_kit/hol/_isabelle_binders.py +125 -0
  171. unicode_logic_kit/hol/classical.py +812 -0
  172. unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
  173. unicode_logic_kit/hol/deepshallow/_common.py +177 -0
  174. unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
  175. unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
  176. unicode_logic_kit/hol/deepshallow/modal.py +217 -0
  177. unicode_logic_kit/hol/deepshallow/qml.py +406 -0
  178. unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
  179. unicode_logic_kit/hol/free.py +753 -0
  180. unicode_logic_kit/hol/goedel.py +336 -0
  181. unicode_logic_kit/hol/ho_modal.py +1743 -0
  182. unicode_logic_kit/hol/intuitionistic.py +403 -0
  183. unicode_logic_kit/hol/isabelle_conditional.py +593 -0
  184. unicode_logic_kit/hol/isabelle_modal.py +1908 -0
  185. unicode_logic_kit/hol/isabelle_relevant.py +412 -0
  186. unicode_logic_kit/hol/isabelle_runner.py +1147 -0
  187. unicode_logic_kit/hol/isabelle_substructural.py +884 -0
  188. unicode_logic_kit/hol/lean.py +1018 -0
  189. unicode_logic_kit/hol/manyvalued.py +921 -0
  190. unicode_logic_kit/hol/secondorder.py +687 -0
  191. unicode_logic_kit/hol/thf_modal.py +941 -0
  192. unicode_logic_kit/hol/thirdorder.py +397 -0
  193. unicode_logic_kit/ilp/__init__.py +89 -0
  194. unicode_logic_kit/ilp/readback.py +389 -0
  195. unicode_logic_kit/ilp/separation.py +153 -0
  196. unicode_logic_kit/ilp/task.py +730 -0
  197. unicode_logic_kit/logic.py +163 -0
  198. unicode_logic_kit/mcp/__init__.py +28 -0
  199. unicode_logic_kit/mcp/__main__.py +5 -0
  200. unicode_logic_kit/mcp/chem_tools.py +1031 -0
  201. unicode_logic_kit/mcp/server.py +2453 -0
  202. unicode_logic_kit/mcp/syntax_spec.py +681 -0
  203. unicode_logic_kit/prob/__init__.py +53 -0
  204. unicode_logic_kit/prob/_bdd.py +225 -0
  205. unicode_logic_kit/prob/_column_gen.py +668 -0
  206. unicode_logic_kit/prob/distribution.py +686 -0
  207. unicode_logic_kit/prob/nilsson.py +470 -0
  208. unicode_logic_kit/py.typed +0 -0
  209. unicode_logic_kit/semantics/__init__.py +137 -0
  210. unicode_logic_kit/semantics/_modal_reject.py +156 -0
  211. unicode_logic_kit/semantics/action_models.py +466 -0
  212. unicode_logic_kit/semantics/asp_models.py +1200 -0
  213. unicode_logic_kit/semantics/conditional.py +580 -0
  214. unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
  215. unicode_logic_kit/semantics/free_logic.py +913 -0
  216. unicode_logic_kit/semantics/fuzzy.py +384 -0
  217. unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
  218. unicode_logic_kit/semantics/intuitionistic.py +581 -0
  219. unicode_logic_kit/semantics/kripke.py +1139 -0
  220. unicode_logic_kit/semantics/manyvalued.py +580 -0
  221. unicode_logic_kit/semantics/matrix.py +342 -0
  222. unicode_logic_kit/semantics/model_eval.py +1135 -0
  223. unicode_logic_kit/semantics/modelfinder.py +1036 -0
  224. unicode_logic_kit/semantics/nonmonotonic.py +372 -0
  225. unicode_logic_kit/semantics/relevant.py +331 -0
  226. unicode_logic_kit/semantics/secondorder.py +657 -0
  227. unicode_logic_kit/semantics/structures.py +352 -0
  228. unicode_logic_kit/semantics/tarski.py +975 -0
  229. unicode_logic_kit/semantics/team.py +315 -0
  230. unicode_logic_kit/semantics/team_translation.py +416 -0
  231. unicode_logic_kit/semantics/thirdorder.py +358 -0
  232. unicode_logic_kit/semantics/tnorm.py +85 -0
  233. unicode_logic_kit/semantics/truthtable.py +201 -0
  234. unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
  235. unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
  236. unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
  237. unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,1637 @@
1
+ """A parser/renderer for the **OWL 2 Manchester Syntax**, restricted to ALCHQ.
2
+
3
+ `OWL 2 Manchester Syntax <https://www.w3.org/TR/owl2-manchester-syntax/>`_ is
4
+ the W3C-standardised, keyword-based (as opposed to :mod:`unicode_logic_kit.dl.parser`'s
5
+ glyph-based) concrete syntax for OWL 2 class expressions — the notation Protégé
6
+ and most OWL tooling show by default (``Person and hasChild some Doctor``
7
+ rather than ``Person ⊓ ∃hasChild.Doctor``). This module is the *OWL bridge,
8
+ part 1*: it parses (and renders) exactly the ALCHQ-expressible fragment of that
9
+ syntax into/from the kit's existing :class:`~unicode_logic_kit.dl.concepts.Concept`
10
+ AST, so any Manchester-syntax ALCHQ expression becomes reachable by every kit
11
+ reasoner and export (:mod:`unicode_logic_kit.dl.tableau`,
12
+ :mod:`unicode_logic_kit.dl.translate`, …) with no separate code path.
13
+
14
+ Supported fragment (ALCHQ, matching :mod:`unicode_logic_kit.dl.concepts` plus the
15
+ role-hierarchy/transitivity RBox axioms below):
16
+
17
+ - class names (``Person``, ``hasChild`` as a role name);
18
+ - ``C and D``, ``C or D``, ``not C``;
19
+ - ``r some C`` (∃r.C), ``r only C`` (∀r.C);
20
+ - ``r value a`` (∃r.{a}) — a value restriction, mapping onto
21
+ :class:`~unicode_logic_kit.dl.concepts.HasValue`, a concept kind of its
22
+ own and NOT ``r some {a}``, which stays refused (see that class's
23
+ docstring for why the two must not collapse into one AST shape);
24
+ - ``r min n C`` / ``r max n C`` / ``r exactly n C`` (≥n r.C / ≤n r.C /
25
+ ≥n r.C ⊓ ≤n r.C) — the qualifying class ``C`` is OPTIONAL per the W3C
26
+ grammar and defaults to ``owl:Thing`` when omitted (``r min n`` alone);
27
+ see "Qualified number restrictions" below;
28
+ - parentheses;
29
+ - ``owl:Thing`` / ``owl:Nothing`` for ⊤ / ⊥ (the W3C grammar treats these
30
+ as ordinary ``owl:``-prefixed class names, not keywords — see "Top and
31
+ Bottom" below — and this module maps exactly those two spellings onto
32
+ :class:`~unicode_logic_kit.dl.concepts.Top` / :class:`~unicode_logic_kit.dl.concepts.Bottom`).
33
+
34
+ Rejected (real Manchester/OWL 2 syntax, but outside ALCHQ — see "Rejected
35
+ constructs" below): ``Self`` restrictions, inverse
36
+ roles (``inverse r``), nominal ("one-of") concepts (``{a, b}``), and the facets
37
+ of a datatype restriction that have no first-order image
38
+ (``xsd:pattern``, the length facets, an ordering facet on a non-numeric base).
39
+ Each is rejected with a
40
+ :class:`ManchesterSyntaxError` naming the specific construct, per the kit's
41
+ honesty convention (an unsupported construct is a loud, precise error —
42
+ never a silent mistranslation or a weaker-than-requested result).
43
+
44
+ Data restrictions
45
+ ------------------
46
+ A restriction whose filler is a DATA RANGE is read as a data restriction
47
+ (:class:`~unicode_logic_kit.dl.concepts.DataExists` and friends), not as an
48
+ object restriction with a class that happens to be called ``xsd:integer``:
49
+
50
+ * ``d some xsd:integer``, ``d only DR`` -> ``DataExists`` / ``DataForAll``;
51
+ * ``d min 2 DR`` / ``d max 2 DR`` / ``d exactly 2 DR`` -> ``DataAtLeast`` /
52
+ ``DataAtMost`` (``exactly`` as their conjunction, as for objects);
53
+ * ``d value "1"^^xsd:integer`` (or a bare numeral, ``d value 1``) ->
54
+ ``DataHasValue``. The BARE numerals read are the W3C grammar's three
55
+ unquoted literal forms: an integer (``1``, ``-3``, an ``xsd:integer``), a
56
+ decimal (``1.5``, an ``xsd:decimal``) and a floating-point literal, which is
57
+ an ``xsd:float`` and ends in ``f`` or ``F`` (``1.5f``, ``2F``, ``-.5f``,
58
+ ``1.0e+2f``). The data layer has no first-order image of ``xsd:float`` (its
59
+ value space has ``NaN`` and a signed zero), so ``d value 1.5f`` is the same
60
+ ``DataHasValue`` as ``d value "1.5"^^xsd:float`` and is refused BY NAME at
61
+ translation, exactly as that quoted spelling and the Functional-Style
62
+ ``"1.5"^^xsd:float`` are; it is never read as an individual called ``1.5f``.
63
+ What is not a literal form stays an individual name: ``1e5`` (no ``f``),
64
+ ``1.5d``, ``1.5fx``;
65
+ * the data ranges: a datatype name, ``DR and DR``, ``DR or DR``, ``not DR``,
66
+ ``{lit, lit}`` and the facet bracket ``xsd:decimal[>= 10000, <= 30000]``
67
+ (``>=``/``<=``/``>``/``<`` for the four ordering facets, with a bare numeral
68
+ typed as the base datatype). Read on its own by
69
+ :func:`parse_manchester_data_range`.
70
+
71
+ Manchester Syntax has no declaration table, so a restriction's property is not
72
+ known to be a data property. The parser needs ONE signal, and the FILLER gives
73
+ it, because OWL 2 forbids a name being both a class and a datatype: a filler is
74
+ a data range when it names a built-in datatype of the OWL 2 datatype map
75
+ (``xsd:integer``, ``rdfs:Literal``, …, either spelling), when it is followed by
76
+ a facet bracket, when it is a brace-enclosed list of literals, or when its name
77
+ is one of the ``datatypes=`` the caller passes. A datatype DEFINED by the
78
+ ontology (``OboRoIdrange1``) has no marker a context-free parse could see:
79
+ ``d some OboRoIdrange1`` reads as an object restriction over a class of that
80
+ name unless ``datatypes=("OboRoIdrange1",)`` says otherwise. ``hasPet some Dog``
81
+ is unchanged. A built-in datatype name in CLASS position is refused by name
82
+ (it cannot be a class), and an unqualified ``d min 2`` has no filler to tell
83
+ by and stays an object restriction.
84
+
85
+ Qualified number restrictions
86
+ ------------------------------
87
+ ``min``/``max``/``exactly`` parse into
88
+ :class:`~unicode_logic_kit.dl.concepts.AtLeast` /
89
+ :class:`~unicode_logic_kit.dl.concepts.AtMost` / their conjunction, per the W3C
90
+ grammar's ``objectPropertyExpression ('min'|'max'|'exactly') nonNegativeInteger
91
+ primary?`` production (`§2.2 <https://www.w3.org/TR/owl2-manchester-syntax/#Class_Expressions>`_):
92
+ the qualifying class is a ``primary`` slot exactly like ``some``/``only``'s
93
+ operand (so it parenthesises/binds by the same precedence rule — see "Grammar
94
+ and precedence" below), but it is OPTIONAL, defaulting to ``owl:Thing`` (⊤)
95
+ when the token right after the integer cannot start one (i.e. is not ``not``,
96
+ ``inverse``, ``{``, ``(``, or a NAME). ``exactly n`` desugars to
97
+ ``AtLeast(n, role, C) ⊓ AtMost(n, role, C)`` at PARSE time (there is no
98
+ separate "exactly" AST node — see :mod:`unicode_logic_kit.dl.concepts`), and
99
+ :func:`to_manchester` renders the two restrictions back out separately (as
100
+ ``role min n ... and role max n ...``), which still round-trips to the exact
101
+ same conjunction rather than needing to be recognised and re-folded into
102
+ ``exactly``. :func:`to_manchester` likewise omits the qualifying-class text
103
+ entirely when it is ⊤ (``role min n``, not ``role min n owl:Thing``), the
104
+ same "unqualified restriction" shorthand real OWL tooling uses; both
105
+ spellings parse back to the identical ``AtLeast``/``AtMost`` with ``Top()``
106
+ as the filler. See :mod:`unicode_logic_kit.dl.tableau`'s "Qualified number
107
+ restrictions" section for the reasoning side (the *simple roles* restriction
108
+ in particular: a number restriction is rejected outright, by role name, if
109
+ the role is transitive or has a transitive sub-role).
110
+
111
+ Grammar and precedence
112
+ -----------------------
113
+ The W3C grammar (`§2.2 <https://www.w3.org/TR/owl2-manchester-syntax/#Class_Expressions>`_)
114
+ is, in its own words, ambiguous "as stated" and resolved by later productions
115
+ binding *more tightly*::
116
+
117
+ description ::= conjunction ('or' conjunction)* -- loosest
118
+ conjunction ::= primary ('and' primary)*
119
+ primary ::= 'not' primary | restriction | atomic
120
+ restriction ::= NAME 'some' primary | NAME 'only' primary | ...
121
+ atomic ::= NAME | '(' description ')' -- tightest
122
+
123
+ i.e. restrictions (``some``/``only``) bind tightest, then ``not``, then
124
+ ``and``, then ``or`` loosest — precisely the precedence lattice
125
+ :mod:`unicode_logic_kit.dl.concepts` already uses for the glyph syntax
126
+ (``_PREC``: ``Or=1 < And=2 < Not=Exists=ForAll=3 < Atomic=4``), so
127
+ :func:`to_manchester` reuses that exact lattice, just spelling the operators
128
+ as keywords instead of glyphs. Two consequences worth spelling out because
129
+ they are easy to get backwards:
130
+
131
+ - ``not`` binds *weaker* than ``some``/``only``: ``not r some A`` parses as
132
+ ``not (r some A)`` (¬∃r.A) — ``primary``'s ``'not' primary`` production
133
+ recurses into a ``primary`` that can itself *be* the whole restriction, so
134
+ ``not`` scopes over it, not over the role name alone (a bare role has no
135
+ negation in ALC in the first place).
136
+ - ``and`` binds tighter than ``or``: ``A and r some B or C`` parses as
137
+ ``(A and (r some B)) or C`` — first ``and`` groups ``A`` with the
138
+ restriction ``r some B`` (restrictions bind tighter than ``and``), then
139
+ the result is ``or``-ed with ``C`` at the loosest level.
140
+ - ``and``/``or`` chains are flat in the grammar (``primary ('and' primary)*``)
141
+ but the AST is binary, so a chain of three or more folds **left**:
142
+ ``A and B and C`` is ``(A and B) and C``, matching
143
+ :mod:`unicode_logic_kit.dl.parser`'s glyph parser exactly. So :func:`to_manchester`
144
+ writes a nested operand of the SAME connective without parentheses on the
145
+ left only: ``And(A, And(B, C))`` is ``A and (B and C)``, never ``A and B and
146
+ C``, which reads back as the other tree.
147
+
148
+ ``and``/``or``/``not``/``some``/``only``/``min``/``max``/``exactly``/
149
+ ``value``/``inverse`` are case-sensitively lowercase keywords; ``Self`` is
150
+ capitalised; ``SubClassOf``/``EquivalentTo`` (used only by
151
+ :func:`parse_manchester_axiom`) are capitalised, with or without a trailing
152
+ colon (``SubClassOf`` and ``SubClassOf:`` are accepted identically — the W3C
153
+ grammar's frame header is colon-terminated, ``SubClassOf:``, but the
154
+ colon-free spelling is the common informal form for a standalone axiom and
155
+ is what callers of this module are expected to write). Per the W3C grammar
156
+ ("Prefixes in abbreviated IRIs must not match any of the keywords of this
157
+ syntax"), none of these words may be used as a bare class or role name.
158
+
159
+ Top and Bottom
160
+ --------------
161
+ The Manchester grammar has no dedicated ⊤/⊥ keyword: ``owl:Thing`` and
162
+ ``owl:Nothing`` are ordinary ``classIRI``s (the prefix ``owl:`` abbreviating
163
+ ``http://www.w3.org/2002/07/owl#``) that merely happen to name the universal
164
+ and empty OWL classes. Since :mod:`unicode_logic_kit.dl.concepts` *does* carry
165
+ first-class :class:`~unicode_logic_kit.dl.concepts.Top`/:class:`~unicode_logic_kit.dl.concepts.Bottom`
166
+ constructors, this module special-cases exactly those two names, in either
167
+ spelling (``owl:Thing`` or the full IRI, bracketed or not; nothing else under the
168
+ ``owl:`` prefix is recognised) rather than leaving them as
169
+ opaque :class:`~unicode_logic_kit.dl.concepts.Atomic` names, so
170
+ ``owl:Thing and C`` reasons exactly like ``⊤ ⊓ C`` under
171
+ :func:`unicode_logic_kit.dl.tableau.concept_satisfiable` and friends.
172
+
173
+ Full IRIs are one name
174
+ ----------------------
175
+ A name in angle brackets with a scheme, ``<http://x.org/a,b>``, is ONE atomic
176
+ name whatever it contains: ``,``, parentheses and square brackets are legal
177
+ inside an IRI, so the tokenizer reads the whole ``<...>`` (and a ``"..."^^<IRI>``
178
+ literal's datatype) before it looks for structural characters. The name is
179
+ STORED WITHOUT its brackets (``Atomic("http://x.org/a,b")``), in every position
180
+ -- a class, a datatype, an object or data property, an individual, a literal's
181
+ datatype, a facet's base -- which is what
182
+ :func:`~unicode_logic_kit.dl.owl_functional.parse_owl_functional` stores too, so
183
+ one IRI is ONE Python string whichever reader produced it: a datatype read here
184
+ matches the literals of that datatype, and ``owl:Thing`` is ``owl:Thing``
185
+ whether it is written ``owl:Thing`` or ``<http://www.w3.org/2002/07/owl#Thing>``
186
+ (the reserved names are compared in their canonical spelling, in every position
187
+ where one of them is special or refused -- a class, a data range, and the
188
+ datatype of a literal). The writers bracket on the way out, and only there:
189
+ :func:`to_manchester` writes a name that is a full IRI (it holds ``://``), or
190
+ that holds a structural character a bare name could not carry, as ``<name>``,
191
+ which reads back as the same string. What tells an IRI from the facet symbols
192
+ ``<`` and ``<=`` is its scheme letter: a facet bound starts with a digit, a sign,
193
+ ``=``, a dot or a quote, never a letter.
194
+
195
+ Names with no spelling
196
+ -----------------------
197
+ A name is written so that the reader reads back THAT name, or it is refused: the
198
+ writer asks the reader's own tokenizer (and, for a class and an individual, its
199
+ parser) whether the spelling reads back as exactly that name, and raises
200
+ :class:`ValueError` naming the name and the reason when no spelling does. The
201
+ bracketed form is the only escape the syntax has, and it needs a scheme and no
202
+ whitespace, so these have none:
203
+
204
+ * a keyword (``and``, ``some``, ``Domain``, ``SubClassOf:``, ...): ``<and>`` has
205
+ no scheme and reads as the name ``<and>``;
206
+ * a name with whitespace, and the empty name (``A and B`` is a conjunction);
207
+ * a name with a structural character ``( ) { } [ ] ,`` and no scheme;
208
+ * a name that already holds the brackets of a full IRI (``<http://x.org/a>``):
209
+ the reader stores an IRI without them, so the text reads back as another name;
210
+ * ``owl:Thing`` and ``owl:Nothing`` as a CLASS, bracketed or not, in either
211
+ spelling: they are the top and the bottom class (as a role or an individual they
212
+ are ordinary names);
213
+ * a built-in datatype (``xsd:integer``, ``rdfs:Literal``, ``owl:real``, ..., in
214
+ either spelling of its namespace, bracketed or not) as a CLASS: the reader
215
+ tells a data restriction from an object restriction by its filler, so
216
+ ``r some xsd:integer`` is a DATA restriction whatever the writer meant, and
217
+ the angle brackets of a full IRI change nothing. As a role, an individual or a
218
+ datatype the name is an ordinary one.
219
+
220
+ A text the reader refuses, as opposed to reads as something else, is not a
221
+ reason to refuse the name.
222
+
223
+ Export-only asymmetry: inverse roles and nominals
224
+ -----------------------------------------------------
225
+ :func:`to_manchester` also RENDERS :class:`~unicode_logic_kit.dl.concepts.InverseRole`
226
+ (as ``inverse r``, in the ``role`` slot of ``some``/``only``/``min``/``max``)
227
+ and :class:`~unicode_logic_kit.dl.concepts.Nominal` (as ``{a}``) — real,
228
+ W3C-legal Manchester syntax the grammar always had room for (see "Rejected
229
+ constructs" above). :func:`parse_manchester` does NOT gain the matching
230
+ read side: ``INVERSE``/``LBRACE`` still hit the same ``_reject`` calls they
231
+ always did, unchanged. This is deliberate, not an oversight: rendering has
232
+ no soundness consequence, so it costs nothing to let a concept built with
233
+ either construct (e.g. from :mod:`unicode_logic_kit.dl.owl_reasoner`'s ALCHQ +
234
+ I/O fragment) still be printed/diffed/logged in this syntax, while parsing it
235
+ back in would re-admit exactly the constructs this module's ALC(HQ)-only
236
+ fragment exists to keep out. ``to_manchester(c)`` followed by
237
+ ``parse_manchester`` on the result therefore round-trips for an
238
+ :class:`~unicode_logic_kit.dl.concepts.InverseRole`/:class:`~unicode_logic_kit.dl.concepts.Nominal`-free
239
+ ``c`` exactly as before, and raises :class:`ManchesterSyntaxError` — naming
240
+ the construct, as always — for one that uses either.
241
+
242
+ Round-trip guarantee
243
+ ---------------------
244
+ ``parse_manchester(to_manchester(c)) == c`` for every concept ``c`` the reader
245
+ reads — every constructor but :class:`~unicode_logic_kit.dl.concepts.Nominal` and
246
+ an :class:`~unicode_logic_kit.dl.concepts.InverseRole` role, which are printed and
247
+ refused by name on the way back — see ``tests/test_owl_manchester.py`` for the
248
+ hand-checked precedence cases and ``tests/test_owl_manchester_roundtrip.py`` for
249
+ every constructor in every child slot and for random nestings. The
250
+ grammar/lattice argument above is why this holds structurally, not just on the
251
+ tested examples: :func:`to_manchester` parenthesises a child exactly when its
252
+ precedence is below the parent slot's threshold — the right operand of ``and``
253
+ (``or``) counting a nested ``and`` (``or``) as below it, because the reader folds
254
+ a flat chain to the left — the same rule :func:`parse_manchester` uses to
255
+ *resolve* precedence when reading text back in.
256
+
257
+ Role axioms: the one-line role-box frames
258
+ --------------------------------------------
259
+ :func:`parse_manchester_role_axiom` reads the role-box axiom shapes
260
+ :class:`~unicode_logic_kit.dl.tableau.TBox` holds — see "Role hierarchies and
261
+ transitive roles (RBox)" and "The rest of the OWL 2 role box" in
262
+ :mod:`unicode_logic_kit.dl.tableau`'s module docstring — each as the one-line
263
+ spelling this module already uses for :func:`parse_manchester_axiom`, rather
264
+ than the full multi-line W3C ``ObjectProperty:`` frame (whose annotation
265
+ slots this kit's role box does not represent). Seven shapes, every frame
266
+ keyword accepted with or without its W3C-grammar trailing colon exactly
267
+ like ``SubClassOf``/``SubClassOf:`` above:
268
+
269
+ * ``"r SubPropertyOf s"`` -> ``("subproperty", "r", "s")``
270
+ * ``"r EquivalentTo s"`` -> ``("equivalentproperty", "r", "s")``
271
+ * ``"r InverseOf s"`` -> ``("inverse", "r", "s")``
272
+ * ``"r DisjointWith s"`` -> ``("disjoint", "r", "s")``
273
+ * ``"r Domain: C"`` -> ``("domain", "r", Concept)``
274
+ * ``"r Range: C"`` -> ``("range", "r", Concept)``
275
+ * ``"r Characteristics: X"`` -> ``("transitive"/"symmetric"/…, "r")``, for
276
+ all SEVEN of OWL 2's object-property characteristics (``Transitive``,
277
+ ``Symmetric``, ``Asymmetric``, ``Reflexive``, ``Irreflexive``,
278
+ ``Functional``, ``InverseFunctional``).
279
+
280
+ ``Domain:``/``Range:`` are the two whose right-hand side is a CLASS
281
+ EXPRESSION rather than a role name, parsed by the same ``_description()``
282
+ every other class-expression position uses (so ``"r Domain: A and B"``
283
+ reads), which is why :func:`parse_manchester_role_axiom`'s return type is
284
+ ``Tuple[object, ...]`` and not ``Tuple[str, ...]``.
285
+
286
+ A PARSER never refuses an axiom KIND — a ``TBox`` is what a parser fills from
287
+ a file, and which kinds the in-house tableau decides is recorded in
288
+ ``dl.tableau._AXIOM_KINDS`` and enforced at query time (see "The axiom-kind
289
+ table" there). So all seven characteristics are read here, and three of them
290
+ (``Symmetric``, ``Reflexive``, ``InverseFunctional``) then make the TABLEAU
291
+ refuse the knowledge base by name while the FOL image still renders them. An
292
+ UNKNOWN characteristic word is still a syntax error naming itself.
293
+
294
+ Two shapes stay refused by name, and the refusal names
295
+ :func:`~unicode_logic_kit.dl.owl_functional.parse_owl_functional` as the entry
296
+ point that does read them:
297
+
298
+ * a PROPERTY CHAIN (``"r o s SubPropertyOf t"``). The bare name ``o`` is
299
+ deliberately NOT promoted to a keyword, so a class or role literally named
300
+ ``o`` keeps working in :func:`parse_manchester`;
301
+ * the comma-separated n-ary ``DisjointWith``/``EquivalentTo`` frame slot.
302
+ ``DisjointWith`` in particular needs ALL pairs, and a one-axiom parser
303
+ returning one pair would quietly produce a weaker theory.
304
+
305
+ One deliberate asymmetry with ``parse_owl_functional``, documented in both: an
306
+ OWL 2 built-in property name (``owl:topObjectProperty`` and the other three)
307
+ is refused here in EVERY position, including the tautological super-role case
308
+ that ``parse_owl_functional`` consumes as a no-op — a single-axiom parser has
309
+ no return shape for "this axiom is nothing", so it names the entry point that
310
+ does. :func:`role_axiom_to_manchester` is the dual renderer for every shape
311
+ above.
312
+ """
313
+
314
+ import functools
315
+ import re
316
+ from typing import FrozenSet, Iterable, List, Optional, Tuple
317
+
318
+ from .concepts import (
319
+ Concept, Top, Bottom, Atomic, Not, And, Or, Exists, ForAll, AtLeast, AtMost,
320
+ InverseRole, Nominal, HasValue, DataExists, DataForAll, DataHasValue,
321
+ DataAtLeast, DataAtMost,
322
+ )
323
+ from .datatypes import (
324
+ BUILTIN_DATATYPES, DataRange, Datatype, DatatypeRestriction, DataOneOf,
325
+ DataComplementOf, DataIntersectionOf, DataUnionOf, Literal,
326
+ UnsupportedDatatypeError, canonical_datatype_name, render_literal_fs,
327
+ EXACT_NUMBER_DATATYPES as _EXACT_BASES,
328
+ )
329
+ from .tableau import (
330
+ RESERVED_TOP_ROLES, RoleExpressionError, _reject_concept_role,
331
+ _reject_concept_roles_deep, reserved_role,
332
+ )
333
+
334
+ __all__ = [
335
+ "parse_manchester", "to_manchester", "parse_manchester_axiom",
336
+ "parse_manchester_role_axiom", "role_axiom_to_manchester",
337
+ "parse_manchester_data_range", "to_manchester_data_range",
338
+ "parse_manchester_literal", "ManchesterSyntaxError",
339
+ ]
340
+
341
+
342
+ class ManchesterSyntaxError(ValueError):
343
+ """Raised by this module's parsers on malformed input *and* on syntax
344
+ that is valid Manchester/OWL 2 but falls outside the ALC fragment (see
345
+ the module docstring's "Rejected constructs"). Always a :class:`ValueError`
346
+ subclass with a message naming the specific offending construct and its
347
+ position in the input, per the kit's honesty convention.
348
+ """
349
+
350
+
351
+ # --------------------------------------------------------------------------- #
352
+ # Tokenizer.
353
+ # --------------------------------------------------------------------------- #
354
+
355
+ # Lowercase connective/restriction keywords (case-sensitive, per the W3C grammar).
356
+ _KEYWORDS = {
357
+ "and": "AND", "or": "OR", "not": "NOT", "some": "SOME", "only": "ONLY",
358
+ "min": "MIN", "max": "MAX", "exactly": "EXACTLY", "value": "VALUE",
359
+ "Self": "SELF", "inverse": "INVERSE",
360
+ }
361
+
362
+ # Axiom-frame keywords, accepted with or without a trailing colon (see module docstring).
363
+ _AXIOM_KEYWORDS = {"SubClassOf": "SUBCLASSOF", "EquivalentTo": "EQUIVALENTTO"}
364
+
365
+ # Role-axiom-frame keywords (see "Role axioms" in the module docstring), same
366
+ # colon-optional convention as _AXIOM_KEYWORDS above. `EquivalentTo` is NOT
367
+ # here: it is already an _AXIOM_KEYWORDS entry (shared with the class-axiom
368
+ # frame) and _classify_word checks that table first, so the role-axiom parser
369
+ # accepts its EQUIVALENTTO token directly.
370
+ _ROLE_AXIOM_KEYWORDS = {
371
+ "SubPropertyOf": "SUBPROPERTYOF",
372
+ "Characteristics": "CHARACTERISTICS",
373
+ "InverseOf": "INVERSEOF",
374
+ "DisjointWith": "DISJOINTWITH",
375
+ "Domain": "DOMAIN",
376
+ "Range": "RANGE",
377
+ }
378
+
379
+ _STRUCT_TOKENS = {
380
+ "(": "LPAREN", ")": "RPAREN",
381
+ "{": "LBRACE", "}": "RBRACE",
382
+ "[": "LBRACKET", "]": "RBRACKET",
383
+ ",": "COMMA",
384
+ }
385
+
386
+ # A Token is (type: str, value: str, pos: int).
387
+ _Token = Tuple[str, str, int]
388
+
389
+ #: A full IRI in angle brackets -- ``<http://x.org/a,b>`` -- which is ONE atomic
390
+ #: name whatever it contains (``,``, parentheses and brackets are all legal inside
391
+ #: an IRI), so the tokenizer reads it before it looks for structural characters.
392
+ #: It must open with a scheme (``letter (letter|digit|+|.|-)* ':'``) and contain
393
+ #: no whitespace; that is what tells it from the facet symbols ``<`` and ``<=``
394
+ #: (``xsd:integer[<5,>=3]``), whose next character is a digit, ``=``, a sign, a
395
+ #: dot or a quote -- never a letter.
396
+ _FULL_IRI = re.compile(r"<[A-Za-z][A-Za-z0-9+.\-]*:[^\s>]*>")
397
+
398
+
399
+ def _unbracket(name: str) -> str:
400
+ """``name`` without the angle brackets of a full IRI: the ONE stored spelling
401
+ of an IRI (see "Full IRIs are one name" in the module docstring). Any other
402
+ name is returned unchanged."""
403
+ return name[1:-1] if _FULL_IRI.fullmatch(name) else name
404
+
405
+
406
+ def _classify_word(word: str) -> str:
407
+ """Classify a maximal non-structural, non-whitespace run as a keyword or NAME."""
408
+ if word in _KEYWORDS:
409
+ return _KEYWORDS[word]
410
+ bare = word[:-1] if word.endswith(":") else word
411
+ if bare in _AXIOM_KEYWORDS:
412
+ return _AXIOM_KEYWORDS[bare]
413
+ if bare in _ROLE_AXIOM_KEYWORDS:
414
+ return _ROLE_AXIOM_KEYWORDS[bare]
415
+ return "NAME"
416
+
417
+
418
+ def _tokenize(text: str) -> List[_Token]:
419
+ """Split ``text`` into structural/keyword/NAME tokens plus a trailing EOF
420
+ sentinel (whose ``pos`` is ``len(text)``, for error messages).
421
+ """
422
+ tokens: List[_Token] = []
423
+ i, n = 0, len(text)
424
+ while i < n:
425
+ ch = text[i]
426
+ if ch.isspace():
427
+ i += 1
428
+ continue
429
+ if ch in _STRUCT_TOKENS:
430
+ tokens.append((_STRUCT_TOKENS[ch], ch, i))
431
+ i += 1
432
+ continue
433
+ if ch == "<":
434
+ iri = _FULL_IRI.match(text, i)
435
+ # ... and it must END there: `<http://x.org/a>b` is one odd word, as
436
+ # it always was, not an IRI followed by a name.
437
+ if iri is not None and (iri.end() == n or text[iri.end()].isspace()
438
+ or text[iri.end()] in _STRUCT_TOKENS):
439
+ tokens.append(("NAME", iri.group(0)[1:-1], i)) # stored WITHOUT brackets
440
+ i = iri.end()
441
+ continue
442
+ if ch == '"':
443
+ # A quoted literal is ONE token, with its ^^datatype or @language
444
+ # suffix: its text may hold whitespace and structural characters
445
+ # (the lexical form of a string), which nothing else here may.
446
+ start = i
447
+ i += 1
448
+ while i < n and text[i] != '"':
449
+ i += 2 if text[i] == "\\" and i + 1 < n else 1
450
+ if i >= n:
451
+ raise ManchesterSyntaxError(
452
+ f"parse_manchester: unterminated string literal starting "
453
+ f"at position {start} in {text!r}")
454
+ i += 1
455
+ iri = _FULL_IRI.match(text, i + 2) if text.startswith("^^", i) else None
456
+ if iri is not None:
457
+ i = iri.end() # "..."^^<IRI>: the datatype IRI is one name
458
+ while i < n and not text[i].isspace() and text[i] not in _STRUCT_TOKENS:
459
+ i += 1
460
+ tokens.append(("LITERAL", text[start:i], start))
461
+ continue
462
+ start = i
463
+ while i < n and not text[i].isspace() and text[i] not in _STRUCT_TOKENS:
464
+ i += 1
465
+ word = text[start:i]
466
+ tokens.append((_classify_word(word), word, start))
467
+ tokens.append(("EOF", "", n))
468
+ return tokens
469
+
470
+
471
+ # --------------------------------------------------------------------------- #
472
+ # Recursive-descent parser (description expressions only; parse_manchester_axiom
473
+ # splits an axiom into two token slices and runs one of these per side).
474
+ # --------------------------------------------------------------------------- #
475
+
476
+ class _Parser:
477
+ """A single parse of one token slice; not re-used across calls."""
478
+
479
+ def __init__(self, tokens: List[_Token], text: str,
480
+ datatypes: Iterable[str] = ()):
481
+ self._tokens = tokens
482
+ self._text = text
483
+ self._i = 0
484
+ self._datatypes: FrozenSet[str] = frozenset(
485
+ canonical_datatype_name(_unbracket(name)) for name in datatypes)
486
+
487
+ def _peek(self) -> _Token:
488
+ return self._tokens[self._i]
489
+
490
+ def _advance(self) -> _Token:
491
+ tok = self._tokens[self._i]
492
+ self._i += 1
493
+ return tok
494
+
495
+ def _error(self, message: str) -> ManchesterSyntaxError:
496
+ return ManchesterSyntaxError(f"{message} in {self._text!r}")
497
+
498
+ def _expect(self, ttype: str, what: str) -> _Token:
499
+ tok = self._peek()
500
+ if tok[0] != ttype:
501
+ found = "end of input" if tok[0] == "EOF" else f"{tok[1]!r}"
502
+ raise self._error(
503
+ f"parse_manchester: expected {what} but found {found} "
504
+ f"at position {tok[2]}")
505
+ return self._advance()
506
+
507
+ def _expect_eof(self) -> None:
508
+ tok = self._peek()
509
+ if tok[0] != "EOF":
510
+ raise self._error(
511
+ f"parse_manchester: unexpected trailing input {tok[1]!r} "
512
+ f"at position {tok[2]}")
513
+
514
+ def _reject(self, what: str, tok: _Token) -> None:
515
+ raise self._error(
516
+ f"parse_manchester: {what} — not supported outside ALC "
517
+ f"(found {tok[1]!r} at position {tok[2]})")
518
+
519
+ def _checked(self, concept: Concept, pos: int) -> Concept:
520
+ """``concept``, unless its role is an OWL 2 built-in property name (or
521
+ ``=``/``≠``) — refused BY NAME, as :func:`parse_manchester_role_axiom`
522
+ and ``dl.parse_owl_functional`` refuse the same name. ``owl:topObject
523
+ Property some A`` used to read as an ordinary role of that name, whose
524
+ verdict is not the universal property's. ``exactly`` is the ``And`` of
525
+ two restrictions over ONE role, so the left one stands for both."""
526
+ try:
527
+ _reject_concept_role(
528
+ concept.left if isinstance(concept, And) else concept,
529
+ where="parse_manchester")
530
+ except RoleExpressionError as exc:
531
+ raise self._error(f"{exc} (at position {pos})") from exc
532
+ return concept
533
+
534
+ # -- grammar levels, loosest first (mirrors the W3C production order) -- #
535
+
536
+ def _description(self) -> Concept:
537
+ left = self._conjunction()
538
+ while self._peek()[0] == "OR":
539
+ self._advance()
540
+ left = Or(left, self._conjunction())
541
+ return left
542
+
543
+ def _conjunction(self) -> Concept:
544
+ left = self._primary()
545
+ while self._peek()[0] == "AND":
546
+ self._advance()
547
+ left = And(left, self._primary())
548
+ return left
549
+
550
+ def _primary(self) -> Concept:
551
+ ttype, value, pos = self._peek()
552
+ if ttype == "NOT":
553
+ self._advance()
554
+ return Not(self._primary())
555
+ if ttype == "INVERSE":
556
+ self._reject("inverse roles ('inverse r')", self._peek())
557
+ if ttype == "LBRACE":
558
+ self._reject("nominal concepts ('{a, b}')", self._peek())
559
+ if ttype == "LPAREN":
560
+ self._advance()
561
+ inner = self._description()
562
+ self._expect("RPAREN", "')'")
563
+ return inner
564
+ if ttype == "NAME":
565
+ self._advance()
566
+ if self._peek()[0] == "LBRACKET":
567
+ raise self._error(
568
+ f"parse_manchester: the facet bracket after {value!r} (at "
569
+ f"position {self._peek()[2]}) makes it a DATA RANGE, which is "
570
+ f"not a class expression. Use it as the filler of a data "
571
+ f"restriction ('d some {value}[...]') or read it on its own "
572
+ f"with parse_manchester_data_range")
573
+ nxt = self._peek()
574
+ if nxt[0] == "SOME":
575
+ self._advance()
576
+ if self._filler_is_data_range():
577
+ return self._checked(DataExists(value, self._data_primary()), pos)
578
+ return self._checked(Exists(value, self._primary()), pos)
579
+ if nxt[0] == "ONLY":
580
+ self._advance()
581
+ if self._filler_is_data_range():
582
+ return self._checked(DataForAll(value, self._data_primary()), pos)
583
+ return self._checked(ForAll(value, self._primary()), pos)
584
+ if nxt[0] in ("MIN", "MAX", "EXACTLY"):
585
+ self._advance()
586
+ return self._checked(self._number_restriction(nxt[0], value), pos)
587
+ if nxt[0] == "VALUE":
588
+ # `r value a` -> HasValue(r, a); `d value "1"^^xsd:integer` (or a
589
+ # bare numeral) -> DataHasValue(d, lit). The keyword was always
590
+ # tokenised (see _KEYWORDS). NOT Exists(value, Nominal(...)):
591
+ # `r some {a}` is the other spelling and stays refused, so the
592
+ # two round-trip apart -- see dl.concepts.HasValue.
593
+ self._advance()
594
+ literal = self._literal_or_none()
595
+ if literal is not None:
596
+ return self._checked(DataHasValue(value, literal), pos)
597
+ return self._checked(HasValue(
598
+ value, self._expect("NAME", "an individual name or a literal")[1]), pos)
599
+ if nxt[0] == "SELF":
600
+ self._reject("Self restrictions ('Self')", nxt)
601
+ canonical = canonical_datatype_name(value)
602
+ if canonical == "owl:Thing":
603
+ return Top()
604
+ if canonical == "owl:Nothing":
605
+ return Bottom()
606
+ if self._is_datatype(value):
607
+ raise self._error(
608
+ f"parse_manchester: {value!r} (at position {pos}) is a "
609
+ f"DATATYPE, not a class — OWL 2 does not let one name be "
610
+ f"both. Use it as the filler of a data restriction "
611
+ f"('d some {value}') or read it with parse_manchester_data_range")
612
+ return Atomic(value)
613
+ found = "end of input" if ttype == "EOF" else f"{value!r}"
614
+ raise self._error(
615
+ f"parse_manchester: unexpected {found} at position {pos}; "
616
+ "expected a class name, 'not', 'owl:Thing', 'owl:Nothing', or '('")
617
+
618
+ # -- qualified number restrictions ('min'/'max'/'exactly') ---------- #
619
+
620
+ _PRIMARY_START = {"NOT", "INVERSE", "LBRACE", "LPAREN", "NAME"}
621
+
622
+ def _number_restriction(self, kind: str, role: str) -> Concept:
623
+ """Parse the ``n primary?`` tail of ``role (min|max|exactly) n primary?``,
624
+ having already consumed ``role`` and the ``kind`` keyword (see the module
625
+ docstring's "Qualified number restrictions"). The qualifying class defaults
626
+ to ``owl:Thing`` (⊤) when the next token cannot start a ``primary``.
627
+ """
628
+ n = self._expect_nonneg_int()
629
+ if self._peek()[0] in self._PRIMARY_START and self._filler_is_data_range():
630
+ datarange = self._data_primary()
631
+ if kind == "MIN":
632
+ return DataAtLeast(n, role, datarange)
633
+ if kind == "MAX":
634
+ return DataAtMost(n, role, datarange)
635
+ return And(DataAtLeast(n, role, datarange), DataAtMost(n, role, datarange))
636
+ filler = self._primary() if self._peek()[0] in self._PRIMARY_START else Top()
637
+ if kind == "MIN":
638
+ return AtLeast(n, role, filler)
639
+ if kind == "MAX":
640
+ return AtMost(n, role, filler)
641
+ return And(AtLeast(n, role, filler), AtMost(n, role, filler)) # 'exactly'
642
+
643
+ def _expect_nonneg_int(self) -> int:
644
+ tok = self._peek()
645
+ if tok[0] == "NAME" and tok[1].isdigit():
646
+ self._advance()
647
+ return int(tok[1])
648
+ found = "end of input" if tok[0] == "EOF" else f"{tok[1]!r}"
649
+ raise self._error(
650
+ f"parse_manchester: expected a non-negative integer but found {found} "
651
+ f"at position {tok[2]}")
652
+
653
+ def parse_description(self) -> Concept:
654
+ c = self._description()
655
+ self._expect_eof()
656
+ return c
657
+
658
+ # -- data ranges and literals ---------------------------------------- #
659
+
660
+ def _is_datatype(self, name: str) -> bool:
661
+ """True iff ``name`` is a datatype for this parse: one of the OWL 2
662
+ datatype map's (either spelling of its namespace) or one the caller
663
+ declared with ``datatypes=``."""
664
+ canonical = canonical_datatype_name(name)
665
+ return canonical in BUILTIN_DATATYPES or canonical in self._datatypes
666
+
667
+ def _filler_is_data_range(self) -> bool:
668
+ """Does the filler that starts at the cursor read as a DATA range?
669
+
670
+ Looks only at the filler's HEAD, skipping any ``not`` and ``(`` in front
671
+ of it: a name that is a datatype or is followed by a facet bracket, or a
672
+ brace-enclosed list that starts with a literal. This is the one signal a
673
+ context-free Manchester parse has (see the module docstring's "Data
674
+ restrictions"); it never consumes anything.
675
+ """
676
+ tokens, i = self._tokens, self._i
677
+ while tokens[i][0] in ("NOT", "LPAREN"):
678
+ i += 1
679
+ ttype, value, _pos = tokens[i]
680
+ if ttype == "NAME":
681
+ return tokens[i + 1][0] == "LBRACKET" or self._is_datatype(value)
682
+ if ttype == "LBRACE":
683
+ head = tokens[i + 1]
684
+ return head[0] == "LITERAL" or (head[0] == "NAME" and _bare_numeral(head[1]))
685
+ return False
686
+
687
+ def _literal_or_none(self) -> Optional[Literal]:
688
+ """Consume a literal if one is at the cursor: a quoted literal token or
689
+ a bare numeral (``1``, ``-3``, ``1.5``, or the floating-point ``1.5f``);
690
+ else ``None``."""
691
+ ttype, value, pos = self._peek()
692
+ if ttype == "LITERAL":
693
+ self._advance()
694
+ return self._build(lambda: _literal_from_token(value, pos), value, pos)
695
+ if ttype == "NAME" and _bare_numeral(value):
696
+ self._advance()
697
+ return self._build(lambda: _bare_literal(value, None), value, pos)
698
+ return None
699
+
700
+ def _build(self, make, what: str, pos: int):
701
+ """``make()``, a data-layer refusal re-raised as the parser's own error
702
+ naming ``what`` and where."""
703
+ try:
704
+ return make()
705
+ except UnsupportedDatatypeError as exc:
706
+ raise self._error(f"parse_manchester: {exc} (at position {pos}, {what!r})") from exc
707
+
708
+ def _data_description(self) -> DataRange:
709
+ """``dataConjunction ('or' dataConjunction)*`` -- an N-ARY union, so a
710
+ chain is one :class:`DataUnionOf`; a parenthesised group stays nested."""
711
+ parts = [self._data_conjunction()]
712
+ while self._peek()[0] == "OR":
713
+ self._advance()
714
+ parts.append(self._data_conjunction())
715
+ return parts[0] if len(parts) == 1 else DataUnionOf(tuple(parts))
716
+
717
+ def _data_conjunction(self) -> DataRange:
718
+ parts = [self._data_primary()]
719
+ while self._peek()[0] == "AND":
720
+ self._advance()
721
+ parts.append(self._data_primary())
722
+ return parts[0] if len(parts) == 1 else DataIntersectionOf(tuple(parts))
723
+
724
+ def _data_primary(self) -> DataRange:
725
+ """``'not'? dataAtomic`` with ``dataAtomic`` a datatype, a facet
726
+ restriction ``dt[facet lit, …]``, a literal set ``{lit, …}`` or a
727
+ parenthesised data range."""
728
+ ttype, value, pos = self._peek()
729
+ if ttype == "NOT":
730
+ self._advance()
731
+ return DataComplementOf(self._data_primary())
732
+ if ttype == "LPAREN":
733
+ self._advance()
734
+ inner = self._data_description()
735
+ self._expect("RPAREN", "')'")
736
+ return inner
737
+ if ttype == "LBRACE":
738
+ self._advance()
739
+ values = [self._expect_literal()]
740
+ while self._peek()[0] == "COMMA":
741
+ self._advance()
742
+ values.append(self._expect_literal())
743
+ self._expect("RBRACE", "'}'")
744
+ return DataOneOf(tuple(values))
745
+ if ttype == "NAME":
746
+ if canonical_datatype_name(value) in ("owl:Thing", "owl:Nothing"):
747
+ # The mirror of the "is a DATATYPE, not a class" refusal in
748
+ # _primary, and the same policy dl.parse_owl_functional has: a
749
+ # class is not a data range, and a datatype called owl:Thing
750
+ # would be an uninterpreted predicate of that name.
751
+ raise self._error(
752
+ f"parse_manchester: {value!r} (at position {pos}) is a CLASS, "
753
+ f"not a data range — OWL 2 does not let one name be both. "
754
+ f"The data domain's own 'everything' is rdfs:Literal; use a "
755
+ f"datatype name here")
756
+ self._advance()
757
+ if self._peek()[0] == "LBRACKET":
758
+ return self._facet_restriction(value, pos)
759
+ return Datatype(value)
760
+ found = "end of input" if ttype == "EOF" else f"{value!r}"
761
+ raise self._error(
762
+ f"parse_manchester: expected a data range but found {found} at "
763
+ f"position {pos}; expected a datatype name, 'not', '{{', or '('")
764
+
765
+ def _expect_literal(self) -> Literal:
766
+ literal = self._literal_or_none()
767
+ if literal is None:
768
+ found = "end of input" if self._peek()[0] == "EOF" else f"{self._peek()[1]!r}"
769
+ raise self._error(
770
+ f"parse_manchester: expected a literal but found {found} at "
771
+ f"position {self._peek()[2]}")
772
+ return literal
773
+
774
+ def _facet_restriction(self, base: str, pos: int) -> DataRange:
775
+ """``dt '[' facet literal (',' facet literal)* ']'`` (the name already
776
+ consumed, the cursor on ``[``)."""
777
+ self._expect("LBRACKET", "'['")
778
+ facets: List[Tuple[str, Literal]] = []
779
+ while True:
780
+ ftype, fvalue, fpos = self._peek()
781
+ if ftype != "NAME":
782
+ found = "end of input" if ftype == "EOF" else f"{fvalue!r}"
783
+ raise self._error(
784
+ f"parse_manchester: expected a facet ({', '.join(_FACET_SYMBOLS)}, "
785
+ f"…) but found {found} at position {fpos}")
786
+ self._advance()
787
+ symbol, glued = fvalue, None
788
+ # `>=5` with no space after the symbol: the symbol is the longest
789
+ # ordering prefix, the rest the bound.
790
+ if fvalue not in _FACET_SYMBOLS:
791
+ for candidate in (">=", "<=", ">", "<"):
792
+ if fvalue.startswith(candidate):
793
+ symbol, glued = candidate, fvalue[len(candidate):]
794
+ break
795
+ if symbol not in _FACET_SYMBOLS:
796
+ raise self._error(
797
+ f"parse_manchester: unknown facet {symbol!r} at position "
798
+ f"{fpos}; the facets are {', '.join(_FACET_SYMBOLS)}")
799
+ if glued is not None:
800
+ bound = self._build(lambda: _bare_literal(glued, base), glued, fpos)
801
+ else:
802
+ btype, bvalue, bpos = self._peek()
803
+ if btype == "LITERAL":
804
+ self._advance()
805
+ bound = self._build(lambda: _literal_from_token(bvalue, bpos), bvalue, bpos)
806
+ elif btype == "NAME" and _bare_numeral(bvalue):
807
+ self._advance()
808
+ bound = self._build(lambda: _bare_literal(bvalue, base), bvalue, bpos)
809
+ else:
810
+ found = "end of input" if btype == "EOF" else f"{bvalue!r}"
811
+ raise self._error(
812
+ f"parse_manchester: expected the facet {symbol!r}'s "
813
+ f"literal but found {found} at position {bpos}")
814
+ facets.append((_FACET_SYMBOLS[symbol], bound))
815
+ if self._peek()[0] == "COMMA":
816
+ self._advance()
817
+ continue
818
+ break
819
+ self._expect("RBRACKET", "']'")
820
+ return self._build(
821
+ lambda: DatatypeRestriction(Datatype(base), tuple(facets)), base, pos)
822
+
823
+ def parse_data_range(self) -> DataRange:
824
+ datarange = self._data_description()
825
+ self._expect_eof()
826
+ return datarange
827
+
828
+
829
+ # --------------------------------------------------------------------------- #
830
+ # Literals and facets.
831
+ # --------------------------------------------------------------------------- #
832
+
833
+ #: Manchester Syntax's facet spellings, mapped to the canonical facet names the
834
+ #: data layer uses. The four ordering symbols are the ones this kit translates;
835
+ #: the keyword facets are read so that the REFUSAL can name them.
836
+ _FACET_SYMBOLS = {
837
+ ">=": "xsd:minInclusive", "<=": "xsd:maxInclusive",
838
+ ">": "xsd:minExclusive", "<": "xsd:maxExclusive",
839
+ "length": "xsd:length", "minLength": "xsd:minLength", "maxLength": "xsd:maxLength",
840
+ "pattern": "xsd:pattern", "langRange": "rdf:langRange",
841
+ "totalDigits": "xsd:totalDigits", "fractionDigits": "xsd:fractionDigits",
842
+ }
843
+ _SYMBOL_OF_FACET = {facet: symbol for symbol, facet in _FACET_SYMBOLS.items()}
844
+
845
+ _BARE_NUMERAL = re.compile(r"[+-]?(?:[0-9]+(?:\.[0-9]+)?|\.[0-9]+)\Z")
846
+ #: The W3C grammar's ``floatingPointLiteral``: a sign, ``digits ['.' digits]`` or
847
+ #: ``'.' digits``, an optional exponent, and the ``f``/``F`` that makes it a
848
+ #: float. Group 1 is the lexical form WITHOUT the suffix (``1.0e+2f`` is the
849
+ #: ``xsd:float`` ``"1.0e+2"``).
850
+ _FLOATING_POINT = re.compile(
851
+ r"([+-]?(?:[0-9]+(?:\.[0-9]+)?|\.[0-9]+)(?:[eE][+-]?[0-9]+)?)[fF]\Z")
852
+ _TYPED_LITERAL = re.compile(r'"((?:[^"\\]|\\.)*)"(?:\^\^(\S+)|@(\S+))?\Z', re.DOTALL)
853
+
854
+
855
+ def _bare_numeral(name: str) -> bool:
856
+ """Is ``name`` a bare Manchester numeral: an integer (``1``, ``-3``), a
857
+ decimal (``1.5``) or a floating-point literal (``1.5f``, ``1.0e+2F``)?"""
858
+ return (_BARE_NUMERAL.match(name) is not None
859
+ or _FLOATING_POINT.match(name) is not None)
860
+
861
+
862
+ def _bare_literal(text: str, base: Optional[str]) -> Literal:
863
+ """The literal a BARE numeral stands for: a floating-point literal is an
864
+ ``xsd:float`` whatever the facet's base (the ``f`` says so, as the quoted
865
+ ``"1.5"^^xsd:float`` does); otherwise typed as the facet's base datatype
866
+ when that is an exact-number datatype (``xsd:decimal[>= 10000]`` bounds are
867
+ decimals), else by its shape — an integer, or a decimal if it has a point."""
868
+ floating = _FLOATING_POINT.match(text)
869
+ if floating is not None:
870
+ return Literal(floating.group(1), "xsd:float")
871
+ canonical = canonical_datatype_name(base) if base is not None else None
872
+ if canonical is not None and canonical in _EXACT_BASES:
873
+ return Literal(text, canonical)
874
+ return Literal(text, "xsd:decimal" if "." in text else "xsd:integer")
875
+
876
+
877
+ def _literal_from_token(token: str, pos: int = 0) -> Literal:
878
+ match = _TYPED_LITERAL.match(token)
879
+ if match is None:
880
+ raise UnsupportedDatatypeError(f"{token!r} is not a literal")
881
+ lexical = re.sub(r'\\(["\\])', r"\1", match.group(1))
882
+ if match.group(3) is not None:
883
+ return Literal(lexical, "xsd:string", match.group(3))
884
+ datatype = match.group(2) or "xsd:string"
885
+ if datatype.startswith("<") and datatype.endswith(">"):
886
+ datatype = datatype[1:-1]
887
+ if canonical_datatype_name(datatype) in ("owl:Thing", "owl:Nothing"):
888
+ # the same refusal a data range gets (see _Parser._data_primary), for the
889
+ # datatype of a literal, in either spelling of the name
890
+ raise UnsupportedDatatypeError(
891
+ f"{datatype!r} is a CLASS, not a data range — OWL 2 does not let "
892
+ f"one name be both, so it cannot be the datatype of a literal. "
893
+ f"The data domain's own 'everything' is rdfs:Literal; use a "
894
+ f"datatype name here")
895
+ return Literal(lexical, datatype)
896
+
897
+
898
+
899
+ # --------------------------------------------------------------------------- #
900
+ # Public API.
901
+ # --------------------------------------------------------------------------- #
902
+
903
+ def parse_manchester_literal(text: str) -> Literal:
904
+ """Parse one Manchester-syntax literal: ``"400"^^xsd:integer``, ``"abc"``
905
+ (an ``xsd:string``), ``"abc"@en``, or a bare numeral (``400`` an
906
+ ``xsd:integer``, ``1.5`` an ``xsd:decimal``, ``1.5f`` an ``xsd:float``).
907
+
908
+ Raises:
909
+ ManchesterSyntaxError: malformed text, an ill-typed literal.
910
+ """
911
+ text = text.strip()
912
+ try:
913
+ if _bare_numeral(text):
914
+ return _bare_literal(text, None)
915
+ return _literal_from_token(text)
916
+ except UnsupportedDatatypeError as exc:
917
+ raise ManchesterSyntaxError(
918
+ f"parse_manchester_literal: {exc} in {text!r}") from exc
919
+
920
+
921
+ def parse_manchester_data_range(text: str, *, datatypes: Iterable[str] = ()) -> DataRange:
922
+ """Parse ``text`` as a Manchester-syntax DATA RANGE: a datatype name,
923
+ ``xsd:decimal[>= 10000, <= 30000]``, ``{1, 2}``, ``not DR``, ``DR and DR``,
924
+ ``DR or DR`` and parentheses (precedence as for class expressions: ``not``
925
+ over ``and`` over ``or``).
926
+
927
+ Args:
928
+ text: The data range.
929
+ datatypes: Extra datatype names, for symmetry with
930
+ :func:`parse_manchester`; a name in a data-range position is a
931
+ datatype whether or not it is listed, so this only matters there.
932
+
933
+ Raises:
934
+ ManchesterSyntaxError: malformed text, or an out-of-scope facet
935
+ (``xsd:pattern``, the length facets, an ordering facet on a
936
+ non-numeric base) -- refused by name.
937
+ """
938
+ return _Parser(_tokenize(text), text, datatypes).parse_data_range()
939
+
940
+
941
+ def to_manchester_data_range(datarange: DataRange) -> str:
942
+ """Render ``datarange`` in Manchester Syntax, dual to
943
+ :func:`parse_manchester_data_range`: ``parse_manchester_data_range(
944
+ to_manchester_data_range(dr)) == dr`` for every data range. A facet bound is
945
+ written as a bare numeral exactly when reading that numeral back gives the
946
+ same literal; otherwise as a typed literal."""
947
+ return _render_datarange(datarange)
948
+
949
+
950
+ def parse_manchester(text: str, *, datatypes: Iterable[str] = ()) -> Concept:
951
+ """Parse ``text`` (OWL 2 Manchester Syntax, ALC fragment) into a :class:`Concept`.
952
+
953
+ Round-trips against :func:`to_manchester`: ``parse_manchester(to_manchester(c))
954
+ == c`` for every ALC concept ``c`` (see the module docstring's "Round-trip
955
+ guarantee").
956
+
957
+ Args:
958
+ text: A Manchester-syntax class expression, e.g.
959
+ ``"Person and hasChild some (Doctor and not Rich)"``.
960
+ datatypes: Extra datatype names to treat as data ranges, beyond the
961
+ OWL 2 datatype map's own. A restriction whose filler names one is a
962
+ DATA restriction (``HasNumber some OboRoIdrange1``); without the
963
+ declaration it reads as an object restriction over a class of that
964
+ name, because Manchester Syntax has no declaration table.
965
+
966
+ Returns:
967
+ The parsed :class:`Concept`.
968
+
969
+ Raises:
970
+ ManchesterSyntaxError: On malformed input (unbalanced parentheses, a
971
+ stray keyword, trailing garbage, …) or on syntax that is valid
972
+ Manchester/OWL 2 but outside ALC (cardinalities, ``value``,
973
+ ``Self``, ``inverse``, nominals, datatype facets — see the module
974
+ docstring's "Rejected constructs").
975
+ """
976
+ return _Parser(_tokenize(text), text, datatypes).parse_description()
977
+
978
+
979
+ def to_manchester(concept: Concept) -> str:
980
+ """Render ``concept`` in OWL 2 Manchester Syntax, dual to :func:`parse_manchester`.
981
+
982
+ Parenthesises a child expression exactly when its precedence is below the
983
+ threshold of the slot it sits in (see the module docstring's "Grammar and
984
+ precedence"; the right operand of ``and`` / ``or`` is parenthesised when it
985
+ is itself an ``and`` / ``or``, since a flat chain reads to the left), so the
986
+ output is minimally parenthesised and re-parses to an identical AST.
987
+
988
+ Args:
989
+ concept: Any ALC :class:`Concept` (as built by
990
+ :mod:`unicode_logic_kit.dl.concepts`'s constructors).
991
+
992
+ Returns:
993
+ The Manchester-syntax rendering, e.g. ``"r some (A and B)"``.
994
+
995
+ Raises:
996
+ ~unicode_logic_kit.dl.tableau.RoleExpressionError:
997
+ a restriction's role (at any depth) is an OWL 2
998
+ built-in property name or ``=`` / ``≠``. :func:`parse_manchester`
999
+ refuses that text, and
1000
+ :func:`~unicode_logic_kit.dl.owl_functional.to_owl_functional_class_expression`
1001
+ refuses the same concept, with the same function — a writer never
1002
+ prints text its own reader refuses.
1003
+ ValueError: a class, role, property or individual name has no spelling
1004
+ that :func:`parse_manchester` reads back as that name (see "Names
1005
+ with no spelling" in the module docstring), or an individual is
1006
+ spelled like a numeral.
1007
+ """
1008
+ _reject_concept_roles_deep(concept, where="to_manchester")
1009
+ return _render(concept)
1010
+
1011
+
1012
+ def parse_manchester_axiom(text: str, *,
1013
+ datatypes: Iterable[str] = ()) -> Tuple[str, Concept, Concept]:
1014
+ """Parse a Manchester-syntax subsumption or equivalence axiom.
1015
+
1016
+ Accepts exactly ``"C SubClassOf D"`` and ``"C EquivalentTo D"`` (the
1017
+ frame keyword may optionally carry its W3C-grammar trailing colon,
1018
+ ``"SubClassOf:"``/``"EquivalentTo:"``), where ``C`` and ``D`` are each
1019
+ parsed by :func:`parse_manchester`. The keyword is located at
1020
+ parenthesis-depth 0; it must occur exactly once.
1021
+
1022
+ Args:
1023
+ text: An axiom of the form ``"<description> SubClassOf <description>"``
1024
+ or ``"<description> EquivalentTo <description>"``.
1025
+
1026
+ Returns:
1027
+ ``("subclass", C, D)`` for ``C SubClassOf D``, or
1028
+ ``("equivalent", C, D)`` for ``C EquivalentTo D``.
1029
+
1030
+ Raises:
1031
+ ManchesterSyntaxError: If no top-level ``SubClassOf``/``EquivalentTo``
1032
+ keyword is found, if more than one is found, or if either side
1033
+ fails to parse as an ALC description (see :func:`parse_manchester`).
1034
+ """
1035
+ tokens = _tokenize(text)
1036
+ depth = 0
1037
+ found: List[Tuple[int, str]] = []
1038
+ for idx, (ttype, _value, _pos) in enumerate(tokens):
1039
+ if ttype == "LPAREN":
1040
+ depth += 1
1041
+ elif ttype == "RPAREN":
1042
+ depth -= 1
1043
+ elif depth == 0 and ttype in ("SUBCLASSOF", "EQUIVALENTTO"):
1044
+ found.append((idx, ttype))
1045
+ if not found:
1046
+ raise ManchesterSyntaxError(
1047
+ "parse_manchester_axiom: expected exactly one top-level "
1048
+ f"'SubClassOf' or 'EquivalentTo' keyword, found none in {text!r}")
1049
+ if len(found) > 1:
1050
+ raise ManchesterSyntaxError(
1051
+ "parse_manchester_axiom: expected exactly one top-level "
1052
+ f"'SubClassOf'/'EquivalentTo' keyword, found {len(found)} in {text!r}")
1053
+ idx, kind = found[0]
1054
+ left_tokens = tokens[:idx] + [("EOF", "", tokens[idx][2])]
1055
+ right_tokens = tokens[idx + 1:]
1056
+ sub = _Parser(left_tokens, text, datatypes).parse_description()
1057
+ sup = _Parser(right_tokens, text, datatypes).parse_description()
1058
+ label = "subclass" if kind == "SUBCLASSOF" else "equivalent"
1059
+ return (label, sub, sup)
1060
+
1061
+
1062
+ # Every OWL 2 object-property characteristic the W3C grammar recognises,
1063
+ # mapped to this module's own tag for it (the first element of
1064
+ # parse_manchester_role_axiom's return tuple). ALL SEVEN are read: a parser
1065
+ # never refuses an axiom KIND -- see "Role axioms" in the module docstring.
1066
+ _CHARACTERISTIC_TAGS = {
1067
+ "Transitive": "transitive",
1068
+ "Symmetric": "symmetric",
1069
+ "Asymmetric": "asymmetric",
1070
+ "Reflexive": "reflexive",
1071
+ "Irreflexive": "irreflexive",
1072
+ "Functional": "functional",
1073
+ "InverseFunctional": "inversefunctional",
1074
+ }
1075
+
1076
+ #: ``tag -> (frame keyword, arity)`` for the four BINARY role-axiom frames
1077
+ #: whose right-hand side is another ROLE NAME.
1078
+ _BINARY_ROLE_FRAMES = {
1079
+ "SUBPROPERTYOF": ("subproperty", "SubPropertyOf"),
1080
+ "EQUIVALENTTO": ("equivalentproperty", "EquivalentTo"),
1081
+ "INVERSEOF": ("inverse", "InverseOf"),
1082
+ "DISJOINTWITH": ("disjoint", "DisjointWith"),
1083
+ }
1084
+
1085
+ #: ``token -> (tag, frame keyword)`` for the two frames whose right-hand side
1086
+ #: is a CLASS EXPRESSION, not a role: ``r Domain: A`` and ``r Range: A``. A
1087
+ #: table of their own because the right side is parsed by ``_description()``
1088
+ #: and the returned tuple carries a :class:`Concept`, which is what widens
1089
+ #: :func:`parse_manchester_role_axiom`'s return type.
1090
+ _FILLER_ROLE_FRAMES = {
1091
+ "DOMAIN": ("domain", "Domain"),
1092
+ "RANGE": ("range", "Range"),
1093
+ }
1094
+
1095
+ _CHAIN_HINT = (
1096
+ "a PROPERTY CHAIN ('r o s SubPropertyOf t') is not one of this module's "
1097
+ "one-line role-axiom shapes: the bare name 'o' is deliberately not a "
1098
+ "keyword here, so a class or role literally named 'o' keeps working in "
1099
+ "parse_manchester. Read a chain with dl.parse_owl_functional "
1100
+ "('SubObjectPropertyOf(ObjectPropertyChain(r s) t)') or build it with "
1101
+ "dl.TBox.add_role_chain")
1102
+
1103
+
1104
+ def _check_role_axiom_name(role: str, where: str) -> None:
1105
+ """Refuse an OWL 2 BUILT-IN property name in a Manchester role axiom.
1106
+
1107
+ Refused in EVERY position, including the tautological super-role case
1108
+ ``dl.parse_owl_functional`` consumes as a no-op: a single-axiom parser has
1109
+ no return shape for "this axiom is nothing". Deliberate asymmetry between
1110
+ the two parsers, documented in both (see "Role axioms" in this module's
1111
+ docstring and "The OWL 2 built-in roles" in ``dl.tableau``'s).
1112
+ """
1113
+ builtin = reserved_role(role)
1114
+ if builtin is None:
1115
+ return
1116
+ universal = builtin in RESERVED_TOP_ROLES
1117
+ raise ManchesterSyntaxError(
1118
+ f"parse_manchester_role_axiom: {role!r} is an OWL 2 BUILT-IN property "
1119
+ f"({builtin} — "
1120
+ + ("the universal property: it relates every pair"
1121
+ if universal else "the empty property: it relates no pair at all")
1122
+ + f"), not an ordinary role name, and ALCHQ (this kit's DL fragment) "
1123
+ f"has neither the universal nor the empty role, so {where} cannot "
1124
+ f"carry it. "
1125
+ + ("An inclusion INTO it is a TAUTOLOGY, which this single-axiom "
1126
+ "parser has no return shape for — dl.parse_owl_functional reads "
1127
+ "that shape and consumes it as a documented no-op."
1128
+ if universal else
1129
+ "'P is empty' is the concept inclusion 'owl:Thing SubClassOf P only "
1130
+ "owl:Nothing', which parse_manchester_axiom does read.")
1131
+ )
1132
+
1133
+
1134
+ def parse_manchester_role_axiom(text: str) -> Tuple[object, ...]:
1135
+ """Parse one Manchester-syntax role-box axiom.
1136
+
1137
+ Seven one-line shapes, of the full W3C ``ObjectProperty:`` frame syntax —
1138
+ see "Role axioms" in the module docstring for why only these, and for the
1139
+ two shapes that stay refused by name. Every frame keyword may carry its
1140
+ W3C-grammar trailing colon (``"SubPropertyOf:"``), exactly like
1141
+ ``SubClassOf``/``SubClassOf:`` in :func:`parse_manchester_axiom`.
1142
+
1143
+ Args:
1144
+ text: A single role axiom, e.g. ``"hasChild SubPropertyOf hasDescendant"``,
1145
+ ``"partOf InverseOf hasPart"``, ``"hasSink DisjointWith hasSource"``,
1146
+ ``"hasSink EquivalentTo hasOutput"``,
1147
+ ``"hasDescendant Characteristics: Transitive"``,
1148
+ ``"Covers Domain: Study"`` or ``"HasUnit Range: Unit and Measurable"``.
1149
+
1150
+ Returns:
1151
+ ``("subproperty", sub_role, super_role)`` (feeding
1152
+ :meth:`~unicode_logic_kit.dl.tableau.TBox.add_role_inclusion`),
1153
+ ``("equivalentproperty", p, q)`` (``add_equivalent_roles``),
1154
+ ``("inverse", p, q)`` (``add_inverse_roles``),
1155
+ ``("disjoint", p, q)`` (``add_disjoint_roles``),
1156
+ ``("domain", role, Concept)`` (``add_role_domain``),
1157
+ ``("range", role, Concept)`` (``add_role_range``), or
1158
+ ``(tag, role)`` for a ``Characteristics:`` declaration, where ``tag``
1159
+ is one of :data:`_CHARACTERISTIC_TAGS`' values and
1160
+ ``"add_" + tag + "_role"`` is the builder — except
1161
+ ``"inversefunctional"``, whose builder is
1162
+ ``add_inverse_functional_role``.
1163
+
1164
+ The return type is ``Tuple[object, ...]``, not ``Tuple[str, ...]``: the
1165
+ ``Domain:``/``Range:`` shapes carry a :class:`Concept` in the third slot,
1166
+ because a domain or range axiom's right-hand side IS a class expression —
1167
+ parsed by the same ``_description()`` every other class-expression position
1168
+ uses, so ``"r Domain: A and B"`` works.
1169
+
1170
+ Raises:
1171
+ ManchesterSyntaxError: malformed input, an unknown role characteristic
1172
+ (named explicitly in the message), an OWL 2 built-in property
1173
+ name, a property chain, or anything not matching one of the seven
1174
+ shapes.
1175
+ """
1176
+ tokens = _tokenize(text)
1177
+
1178
+ def error(message: str) -> ManchesterSyntaxError:
1179
+ return ManchesterSyntaxError(f"parse_manchester_role_axiom: {message} in {text!r}")
1180
+
1181
+ if tokens[0][0] != "NAME":
1182
+ found = "end of input" if tokens[0][0] == "EOF" else f"{tokens[0][1]!r}"
1183
+ raise error(f"expected a role name but found {found} at position {tokens[0][2]}")
1184
+ role = tokens[0][1]
1185
+ keyword = tokens[1]
1186
+
1187
+ if keyword[0] in _BINARY_ROLE_FRAMES:
1188
+ tag, spelling = _BINARY_ROLE_FRAMES[keyword[0]]
1189
+ if len(tokens) == 4 and tokens[2][0] == "NAME" and tokens[3][0] == "EOF":
1190
+ _check_role_axiom_name(role, f"a {spelling} axiom")
1191
+ _check_role_axiom_name(tokens[2][1], f"a {spelling} axiom")
1192
+ return (tag, role, tokens[2][1])
1193
+ if (keyword[0] in ("DISJOINTWITH", "EQUIVALENTTO")
1194
+ and any(t[0] == "NAME" for t in tokens[2:])
1195
+ and text.count(",") > 0):
1196
+ raise error(
1197
+ f"the comma-separated n-ary {spelling} frame slot is not one "
1198
+ f"of this module's one-line role-axiom shapes — only the "
1199
+ f"binary '<role> {spelling} <role>' is. Read the n-ary form "
1200
+ f"with dl.parse_owl_functional, which expands it correctly "
1201
+ f"(DisjointObjectProperties needs ALL pairs, not a chain of "
1202
+ f"consecutive ones), or call dl.TBox."
1203
+ + ("add_disjoint_roles" if keyword[0] == "DISJOINTWITH"
1204
+ else "add_equivalent_roles")
1205
+ + " with every role")
1206
+ raise error(f"expected exactly '<role> {spelling} <role>'")
1207
+
1208
+ if keyword[0] in _FILLER_ROLE_FRAMES:
1209
+ tag, spelling = _FILLER_ROLE_FRAMES[keyword[0]]
1210
+ _check_role_axiom_name(role, f"a {spelling}: axiom")
1211
+ if tokens[2][0] == "EOF":
1212
+ raise error(f"expected exactly '<role> {spelling}: <description>'")
1213
+ # The filler goes through the full description grammar, so
1214
+ # `r Domain: A and B` and `r Range: s some C` both read -- a domain or
1215
+ # range axiom's right-hand side is a CLASS EXPRESSION, not a name.
1216
+ filler = _Parser(list(tokens[2:]), text).parse_description()
1217
+ return (tag, role, filler)
1218
+
1219
+ if keyword[0] == "CHARACTERISTICS":
1220
+ if len(tokens) == 4 and tokens[2][0] == "NAME" and tokens[3][0] == "EOF":
1221
+ characteristic = tokens[2][1]
1222
+ if characteristic in _CHARACTERISTIC_TAGS:
1223
+ _check_role_axiom_name(role, f"a Characteristics: "
1224
+ f"{characteristic} axiom")
1225
+ return (_CHARACTERISTIC_TAGS[characteristic], role)
1226
+ raise error(
1227
+ f"unknown role characteristic {characteristic!r} — OWL 2 has "
1228
+ f"exactly seven: " + ", ".join(sorted(_CHARACTERISTIC_TAGS)))
1229
+ raise error("expected exactly '<role> Characteristics: <characteristic>'")
1230
+
1231
+ if keyword[0] == "NAME" and keyword[1] == "o":
1232
+ raise error(_CHAIN_HINT)
1233
+
1234
+ found = "end of input" if keyword[0] == "EOF" else f"{keyword[1]!r}"
1235
+ raise error(
1236
+ f"expected 'SubPropertyOf', 'EquivalentTo', 'InverseOf', "
1237
+ f"'DisjointWith', 'Domain:', 'Range:' or 'Characteristics:' but found "
1238
+ f"{found} at position {keyword[2]}")
1239
+
1240
+
1241
+ # --------------------------------------------------------------------------- #
1242
+ # Renderer.
1243
+ # --------------------------------------------------------------------------- #
1244
+
1245
+ # Same lattice as concepts.py's _PREC (Or=1 < And=2 < Not=Exists=ForAll=AtLeast=
1246
+ # AtMost=3 < Atomic=4): see the module docstring's "Grammar and precedence" for why
1247
+ # the two coincide.
1248
+ _PREC = {Or: 1, And: 2, Not: 3, Exists: 3, ForAll: 3, AtLeast: 3, AtMost: 3,
1249
+ HasValue: 3, DataExists: 3, DataForAll: 3, DataHasValue: 3,
1250
+ DataAtLeast: 3, DataAtMost: 3,
1251
+ Atomic: 4, Top: 4, Bottom: 4, Nominal: 4}
1252
+
1253
+
1254
+ def _one_name_token(spelled: str, name: str) -> Optional[List[_Token]]:
1255
+ """The tokens of ``spelled`` when the reader takes it for ONE plain name that
1256
+ is exactly ``name`` (not a keyword, not a literal, not several words), else
1257
+ ``None``."""
1258
+ try:
1259
+ tokens = _tokenize(spelled)
1260
+ except ManchesterSyntaxError:
1261
+ return None
1262
+ if len(tokens) == 2 and tokens[0][0] == "NAME" and tokens[0][1] == name:
1263
+ return tokens
1264
+ return None
1265
+
1266
+
1267
+ def _why_no_spelling(name: str) -> str:
1268
+ """Why no spelling of ``name`` is read back as that one name."""
1269
+ if name == "":
1270
+ return "it is empty"
1271
+ if _FULL_IRI.fullmatch(name):
1272
+ return ("it already holds the angle brackets of a full IRI, and the reader "
1273
+ "stores an IRI without them, so <...> reads back as another name "
1274
+ "(the text between the brackets)")
1275
+ if any(ch.isspace() for ch in name):
1276
+ return ("it holds whitespace, where the reader splits a name, and only a "
1277
+ "full IRI (a scheme, no whitespace) is written in angle brackets")
1278
+ if _classify_word(name) != "NAME":
1279
+ return ("it is a keyword of this syntax, which has no escape for it: only "
1280
+ "a full IRI (a scheme, no whitespace) is written in angle brackets")
1281
+ if any(ch in _STRUCT_TOKENS for ch in name):
1282
+ return ("it holds one of ( ) { } [ ] , and has no scheme to be bracketed "
1283
+ "with as a full IRI")
1284
+ return "no spelling of it is read back as one name"
1285
+
1286
+
1287
+ def _names_a_builtin_datatype(name: str) -> bool:
1288
+ """True iff ``name`` is a built-in datatype of the OWL 2 datatype map, in
1289
+ either spelling of its namespace (``xsd:integer`` or the full IRI, bracketed
1290
+ or not) — what the reader takes for a datatype wherever a datatype can stand."""
1291
+ return canonical_datatype_name(name) in BUILTIN_DATATYPES
1292
+
1293
+
1294
+ _DATATYPE_CLASS_REASON = (
1295
+ "it is the name of a built-in datatype, and the reader reads that name as the "
1296
+ "datatype in every spelling (the full IRI and either namespace form included): "
1297
+ "after some / only / min / max / exactly it turns the restriction into a DATA "
1298
+ "restriction, and in any other position it refuses the text, because OWL 2 "
1299
+ "does not let one name be both a class and a datatype")
1300
+
1301
+
1302
+ @functools.lru_cache(maxsize=8192)
1303
+ def _name_spelling(name: str, kind: str) -> Tuple[Optional[str], str]:
1304
+ """``(spelling, "")`` for the one-token spelling of ``name`` that the reader
1305
+ reads back as exactly that name, or ``(None, reason)`` when there is none.
1306
+
1307
+ The spellings tried are the full IRI ``<name>`` (first, for a name that holds
1308
+ ``://`` or a structural character, as before) and the bare name; a spelling
1309
+ counts only when the READER's own tokenizer takes it for one plain name, and,
1310
+ for a ``kind`` of ``"class"`` / ``"individual"``, when the reader's own
1311
+ parser then reads it back as that class / that individual and not as another
1312
+ expression (``owl:Thing`` is the top class in every spelling; a numeral is a
1313
+ data value).
1314
+
1315
+ A class named like a built-in datatype has no spelling at all. The reader
1316
+ decides between an object and a data restriction by the FILLER, so
1317
+ ``r some xsd:integer`` is a data restriction whatever the writer meant, and
1318
+ the full IRI reads the same way: refusing the bare class alone, which the
1319
+ reader refuses on its own, would leave the same name written, and read as
1320
+ something else, in the one position where it is a filler.
1321
+ """
1322
+ candidates: List[str] = []
1323
+ bracketed = f"<{name}>"
1324
+ if ("://" in name or any(ch in _STRUCT_TOKENS for ch in name)) \
1325
+ and _FULL_IRI.fullmatch(bracketed):
1326
+ candidates.append(bracketed)
1327
+ candidates.append(name)
1328
+ for spelled in candidates:
1329
+ tokens = _one_name_token(spelled, name)
1330
+ if tokens is None:
1331
+ continue
1332
+ if kind == "class":
1333
+ if _names_a_builtin_datatype(name):
1334
+ return None, _DATATYPE_CLASS_REASON
1335
+ try:
1336
+ back = _Parser(tokens, spelled).parse_description()
1337
+ except ManchesterSyntaxError:
1338
+ return spelled, "" # refused on reading: loud, not another reading
1339
+ if back != Atomic(name):
1340
+ return None, (f"read as a class it is {back!r}, in every spelling "
1341
+ f"(owl:Thing and owl:Nothing are the top and the bottom class)")
1342
+ elif kind == "individual":
1343
+ text = f"r value {spelled}"
1344
+ try:
1345
+ back = _Parser(_tokenize(text), text).parse_description()
1346
+ except ManchesterSyntaxError:
1347
+ return spelled, ""
1348
+ if back != HasValue("r", name):
1349
+ return None, f"read after 'value' it is {back!r}, not an individual"
1350
+ return spelled, ""
1351
+ return None, _why_no_spelling(name)
1352
+
1353
+
1354
+ def _render_name(name: str, kind: str = "name") -> str:
1355
+ """A kit name as ONE Manchester token that reads back as that name: ``<name>``
1356
+ when it is a full IRI (it holds ``://``) or holds a structural character a
1357
+ bare name could not carry, the bare name otherwise. The brackets are added
1358
+ here and only here (the readers store an IRI without them).
1359
+
1360
+ ``kind`` is ``"class"`` or ``"individual"`` where the reader gives the name a
1361
+ meaning of its own (``owl:Thing``, a numeral), and ``"name"`` elsewhere.
1362
+
1363
+ Raises:
1364
+ ValueError: no spelling reads back as ``name``: it is a keyword of the
1365
+ syntax, holds whitespace or a structural character that no full IRI
1366
+ can carry (an IRI needs a scheme and no whitespace), is empty, already
1367
+ holds the brackets of a full IRI (which the reader strips), is
1368
+ read as another expression (``owl:Thing``), or is a built-in
1369
+ datatype's name used as a class (the reader makes a restriction
1370
+ over it a data restriction). The text is never written as
1371
+ something that reads back as another name or expression.
1372
+ """
1373
+ spelling, reason = _name_spelling(name, kind)
1374
+ if spelling is None:
1375
+ if reason == _DATATYPE_CLASS_REASON:
1376
+ remedy = ("Rename the class: every reader of this kit reads a "
1377
+ "built-in datatype's name as that datatype, or refuses it as "
1378
+ "a class.")
1379
+ else:
1380
+ remedy = ("Rename it, or write the concept with "
1381
+ "dl.to_owl_functional_class_expression, whose <...> carries a "
1382
+ "name that has no '>' in it.")
1383
+ raise ValueError(
1384
+ f"to_manchester: the name {name!r} cannot be written so that "
1385
+ f"parse_manchester reads it back as that name: {reason}. {remedy}")
1386
+ return spelling
1387
+
1388
+
1389
+ def _render_literal(literal: Literal) -> str:
1390
+ """A literal as one Manchester token: ``"5"^^xsd:integer`` with the datatype
1391
+ written through :func:`_render_name`, so a user datatype IRI is bracketed
1392
+ (``"5"^^<http://ex.org/dt,Small>``) and reads back as the same datatype."""
1393
+ return render_literal_fs(literal, _render_name)
1394
+
1395
+
1396
+ def _render_role(role) -> str:
1397
+ """Render a ``role`` field: ``inverse r`` for an
1398
+ :class:`~unicode_logic_kit.dl.concepts.InverseRole` (the W3C grammar's own
1399
+ spelling — see "Export-only asymmetry" in the module docstring), the bare
1400
+ name otherwise.
1401
+ """
1402
+ if isinstance(role, InverseRole):
1403
+ return f"inverse {_render_name(role.role)}"
1404
+ return _render_name(role)
1405
+
1406
+
1407
+ def _render(c: Concept) -> str:
1408
+ """Render a concept with precedence-aware parenthesisation."""
1409
+ if isinstance(c, Top):
1410
+ return "owl:Thing"
1411
+ if isinstance(c, Bottom):
1412
+ return "owl:Nothing"
1413
+ if isinstance(c, Atomic):
1414
+ return _render_name(c.name, "class")
1415
+ if isinstance(c, Nominal):
1416
+ # Behind `some` / `only` / `min` / `max` the reader takes `{3}` for a set of
1417
+ # DATA values, not for a nominal, so an individual spelled like a numeral
1418
+ # has no nominal spelling that reads back as it (cf. the value restriction).
1419
+ if _bare_numeral(c.individual):
1420
+ raise ValueError(
1421
+ f"to_manchester: the individual {c.individual!r} of the nominal "
1422
+ f"{c.to_unicode()} is spelled like a numeral, and Manchester Syntax "
1423
+ f"reads '{{{c.individual}}}' behind a role keyword as a set of DATA "
1424
+ f"values. There is no spelling that reads back as this individual: "
1425
+ f"rename it, or write the concept with "
1426
+ f"dl.to_owl_functional_class_expression.")
1427
+ return "{" + _render_name(c.individual) + "}"
1428
+ if isinstance(c, Not):
1429
+ return "not " + _paren(c.concept, 3)
1430
+ # The reader folds a flat chain LEFT (``A and B and C`` is ``(A and B) and C``),
1431
+ # so only a LEFT operand of the same connective may go unparenthesised: the
1432
+ # right operand is written one level tighter, and an ``and`` inside an
1433
+ # ``and`` (an ``or`` inside an ``or``) on the right keeps its parentheses.
1434
+ if isinstance(c, And):
1435
+ return f"{_paren(c.left, 2)} and {_paren(c.right, 3)}"
1436
+ if isinstance(c, Or):
1437
+ return f"{_paren(c.left, 1)} or {_paren(c.right, 2)}"
1438
+ if isinstance(c, Exists):
1439
+ return f"{_render_role(c.role)} some {_paren(c.concept, 3)}"
1440
+ if isinstance(c, ForAll):
1441
+ return f"{_render_role(c.role)} only {_paren(c.concept, 3)}"
1442
+ if isinstance(c, HasValue):
1443
+ # `r value 5` and `r value 1.5f` are DATA value restrictions to the
1444
+ # reader: an individual whose name is spelled like a bare numeral cannot
1445
+ # be written so that it reads back as an individual. Refused by name,
1446
+ # like every other text this writer would not read back as it was.
1447
+ if _bare_numeral(c.individual):
1448
+ raise ValueError(
1449
+ f"to_manchester: the individual {c.individual!r} of the value "
1450
+ f"restriction {c.to_unicode()} is spelled like a numeral, and "
1451
+ f"Manchester Syntax reads '{_render_role(c.role)} value "
1452
+ f"{c.individual}' as a DATA value restriction to that literal. "
1453
+ f"There is no spelling that reads back as this individual: "
1454
+ f"rename it, or write the axiom with dl.to_owl_functional.")
1455
+ # No _paren: the operand is an individual NAME, never a nested
1456
+ # description, so there is no precedence question to ask.
1457
+ return f"{_render_role(c.role)} value {_render_name(c.individual, 'individual')}"
1458
+ if isinstance(c, AtLeast):
1459
+ return _number_restriction_render(c.role, "min", c.n, c.concept)
1460
+ if isinstance(c, AtMost):
1461
+ return _number_restriction_render(c.role, "max", c.n, c.concept)
1462
+ # The data restrictions. The filler of a data cardinality is ALWAYS written
1463
+ # (even rdfs:Literal): `d min 2` alone reads back as an OBJECT restriction,
1464
+ # because with no filler there is nothing to say it is a data property.
1465
+ if isinstance(c, DataExists):
1466
+ return f"{_render_name(c.prop)} some {_render_datarange(c.datarange, paren=True)}"
1467
+ if isinstance(c, DataForAll):
1468
+ return f"{_render_name(c.prop)} only {_render_datarange(c.datarange, paren=True)}"
1469
+ if isinstance(c, DataHasValue):
1470
+ return f"{_render_name(c.prop)} value {_render_literal(c.value)}"
1471
+ if isinstance(c, DataAtLeast):
1472
+ return f"{_render_name(c.prop)} min {c.n} {_render_datarange(c.datarange, paren=True)}"
1473
+ if isinstance(c, DataAtMost):
1474
+ return f"{_render_name(c.prop)} max {c.n} {_render_datarange(c.datarange, paren=True)}"
1475
+ raise TypeError(f"to_manchester: unsupported concept {type(c).__name__}")
1476
+
1477
+
1478
+ def _render_bound(bound: Literal, base: str) -> str:
1479
+ """A facet bound as a bare numeral if reading it back gives the same
1480
+ literal, else as a typed literal."""
1481
+ bare = bound.lexical.strip()
1482
+ if _bare_numeral(bare):
1483
+ try:
1484
+ if _bare_literal(bare, base) == bound:
1485
+ return bare
1486
+ except UnsupportedDatatypeError:
1487
+ pass
1488
+ return _render_literal(bound)
1489
+
1490
+
1491
+ def _render_datarange(dr: DataRange, *, paren: bool = False) -> str:
1492
+ """Render a data range; ``paren`` parenthesises a compound one, for a slot
1493
+ (a restriction's filler, an ``and``/``or`` operand) that binds tighter."""
1494
+ if isinstance(dr, Datatype):
1495
+ return _render_name(dr.name)
1496
+ if isinstance(dr, DatatypeRestriction):
1497
+ facets = ", ".join(f"{_SYMBOL_OF_FACET[facet]} {_render_bound(bound, dr.base.name)}"
1498
+ for facet, bound in dr.facets)
1499
+ return f"{_render_name(dr.base.name)}[{facets}]"
1500
+ if isinstance(dr, DataOneOf):
1501
+ return "{" + ", ".join(_render_bound(v, "") for v in dr.values) + "}"
1502
+ if isinstance(dr, DataComplementOf):
1503
+ return "not " + _render_datarange(dr.datarange, paren=True)
1504
+ if isinstance(dr, DataIntersectionOf):
1505
+ inner = " and ".join(_render_datarange(r, paren=True) for r in dr.ranges)
1506
+ return f"({inner})" if paren else inner
1507
+ if isinstance(dr, DataUnionOf):
1508
+ inner = " or ".join(_render_datarange(r, paren=True) for r in dr.ranges)
1509
+ return f"({inner})" if paren else inner
1510
+ raise TypeError(f"to_manchester: unsupported data range {type(dr).__name__}")
1511
+
1512
+
1513
+ def _number_restriction_render(role, keyword: str, n: int, filler: Concept) -> str:
1514
+ """Render ``role (min|max) n filler``, omitting the qualifying class entirely
1515
+ when ``filler`` is ⊤ (the "unqualified restriction" shorthand — see the module
1516
+ docstring's "Qualified number restrictions"; both spellings parse back to the
1517
+ same ``Top()``-qualified concept).
1518
+ """
1519
+ rendered_role = _render_role(role)
1520
+ if isinstance(filler, Top):
1521
+ return f"{rendered_role} {keyword} {n}"
1522
+ return f"{rendered_role} {keyword} {n} {_paren(filler, 3)}"
1523
+
1524
+
1525
+ def _paren(c: Concept, parent_prec: int) -> str:
1526
+ """Parenthesise ``c`` when its precedence is below the parent slot's threshold."""
1527
+ inner = _render(c)
1528
+ return f"({inner})" if _PREC.get(type(c), 4) < parent_prec else inner
1529
+
1530
+
1531
+ #: ``tag -> frame keyword`` for the four binary shapes, derived from
1532
+ #: :data:`_BINARY_ROLE_FRAMES` so the reader and the writer cannot name the
1533
+ #: same shape differently.
1534
+ _BINARY_FRAME_SPELLING = {tag: spelling
1535
+ for tag, spelling in _BINARY_ROLE_FRAMES.values()}
1536
+
1537
+ #: ``tag -> frame keyword`` for the two CLASS-EXPRESSION frames, derived from
1538
+ #: :data:`_FILLER_ROLE_FRAMES` for the same reason.
1539
+ _FILLER_FRAME_SPELLING = {tag: spelling
1540
+ for tag, spelling in _FILLER_ROLE_FRAMES.values()}
1541
+
1542
+ #: ``tag -> characteristic word``, the inverse of :data:`_CHARACTERISTIC_TAGS`.
1543
+ _TAG_CHARACTERISTIC = {tag: word for word, tag in _CHARACTERISTIC_TAGS.items()}
1544
+
1545
+
1546
+ def _renderable_role(role: object, axiom: Tuple[object, ...]) -> str:
1547
+ """The role operand of a role axiom, or a ``ValueError`` naming why not.
1548
+
1549
+ The reader takes a plain role NAME in every position (an inverse role has
1550
+ no one-line spelling here, and the OWL 2 built-ins are refused by name), so
1551
+ writing anything else would print text that does not read back -- for an
1552
+ ``InverseRole`` the dataclass ``repr``, ``InverseRole(role='s')``, which
1553
+ no reader would take for a role.
1554
+ """
1555
+ if not isinstance(role, str):
1556
+ raise ValueError(
1557
+ f"role_axiom_to_manchester: a role operand must be a role NAME (a "
1558
+ f"str), got {role!r} in {axiom!r}. The one-line role-axiom syntax "
1559
+ f"has no spelling for an inverse role (dl.parse_manchester_role_axiom "
1560
+ f"reads none): write it with dl.to_owl_functional, which spells a "
1561
+ f"role inclusion with an inverse as SubObjectPropertyOf(r "
1562
+ f"ObjectInverseOf(s)).")
1563
+ builtin = reserved_role(role)
1564
+ if builtin is not None:
1565
+ raise ValueError(
1566
+ f"role_axiom_to_manchester: {role!r} is an OWL 2 BUILT-IN property "
1567
+ f"({builtin}), not an ordinary role name -- "
1568
+ f"dl.parse_manchester_role_axiom refuses it in every position, so "
1569
+ f"the text written here would not read back, in {axiom!r}.")
1570
+ return _render_name(role)
1571
+
1572
+
1573
+ def role_axiom_to_manchester(*axiom: object) -> str:
1574
+ """Render a role-box axiom, dual to :func:`parse_manchester_role_axiom`.
1575
+
1576
+ Round-trips: ``parse_manchester_role_axiom(role_axiom_to_manchester(*axiom))
1577
+ == axiom`` for every ``axiom`` one of that function's return shapes. All
1578
+ three tables are derived from the reader's own (see
1579
+ :data:`_BINARY_FRAME_SPELLING`), so a shape the reader gains cannot be
1580
+ spelled differently here.
1581
+
1582
+ Args:
1583
+ *axiom: One of ``("subproperty", sub, sup)``,
1584
+ ``("equivalentproperty", p, q)``, ``("inverse", p, q)``,
1585
+ ``("disjoint", p, q)``, ``("domain", role, Concept)``,
1586
+ ``("range", role, Concept)`` or ``(characteristic_tag, role)`` —
1587
+ exactly what :func:`parse_manchester_role_axiom` returns.
1588
+
1589
+ Returns:
1590
+ The one-line Manchester spelling, e.g.
1591
+ ``"partOf InverseOf hasPart"``, ``"r Characteristics: Asymmetric"`` or
1592
+ ``"Covers Domain: Study"``.
1593
+
1594
+ Raises:
1595
+ ValueError: ``axiom`` is not one of the recognised shapes, or a role
1596
+ operand is not a plain role name (an ``InverseRole``, or an OWL 2
1597
+ built-in property name, which the reader refuses in every position
1598
+ -- text it would not read back is never written), or the class
1599
+ expression of a ``Domain:`` / ``Range:`` axiom has one as the role
1600
+ of a restriction at any depth (a
1601
+ :class:`~unicode_logic_kit.dl.tableau.RoleExpressionError`, as for
1602
+ :func:`to_manchester`).
1603
+ """
1604
+ # The tag is narrowed to `str` before any table lookup: the parameter is
1605
+ # `object` because the Domain:/Range: shapes carry a Concept, and a lookup
1606
+ # keyed on an un-narrowed `object` is exactly the kind of thing the pinned
1607
+ # mypy configuration refuses.
1608
+ tag = axiom[0] if axiom else None
1609
+ if not isinstance(tag, str):
1610
+ raise ValueError(
1611
+ f"role_axiom_to_manchester: the first element must be the shape's "
1612
+ f"TAG (a str), got {tag!r} in {axiom!r}")
1613
+ if len(axiom) == 3 and tag in _FILLER_FRAME_SPELLING:
1614
+ filler = axiom[2]
1615
+ if not isinstance(filler, Concept):
1616
+ raise ValueError(
1617
+ f"role_axiom_to_manchester: a {tag!r} axiom's third element is "
1618
+ f"its CLASS EXPRESSION, got {filler!r}")
1619
+ # The keyword keeps its colon here, unlike the role-to-role frames: the
1620
+ # W3C grammar's frame slot is `Domain: <description>` and the colon is
1621
+ # what tells a reader the right-hand side is a class expression (the
1622
+ # reader accepts it with or without, like every other frame keyword).
1623
+ role = _renderable_role(axiom[1], axiom)
1624
+ _reject_concept_roles_deep(filler, where="role_axiom_to_manchester")
1625
+ return f"{role} {_FILLER_FRAME_SPELLING[tag]}: {_render(filler)}"
1626
+ if len(axiom) == 3 and tag in _BINARY_FRAME_SPELLING:
1627
+ first = _renderable_role(axiom[1], axiom)
1628
+ second = _renderable_role(axiom[2], axiom)
1629
+ return f"{first} {_BINARY_FRAME_SPELLING[tag]} {second}"
1630
+ if len(axiom) == 2 and tag in _TAG_CHARACTERISTIC:
1631
+ role = _renderable_role(axiom[1], axiom)
1632
+ return f"{role} Characteristics: {_TAG_CHARACTERISTIC[tag]}"
1633
+ raise ValueError(
1634
+ f"role_axiom_to_manchester: expected one of "
1635
+ f"{sorted(_BINARY_FRAME_SPELLING)} with two roles, one of "
1636
+ f"{sorted(_FILLER_FRAME_SPELLING)} with a role and a concept, or one "
1637
+ f"of {sorted(_TAG_CHARACTERISTIC)} with one role, got {axiom!r}")