unicode-logic-kit 0.31.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. unicode_logic_kit/__init__.py +385 -0
  2. unicode_logic_kit/__main__.py +520 -0
  3. unicode_logic_kit/_deadline.py +219 -0
  4. unicode_logic_kit/ace/__init__.py +126 -0
  5. unicode_logic_kit/ace/_align.py +135 -0
  6. unicode_logic_kit/ace/chem_lexicon.py +128 -0
  7. unicode_logic_kit/ace/drs_reader.py +570 -0
  8. unicode_logic_kit/ace/mapping.py +666 -0
  9. unicode_logic_kit/ace/reverse_modal.py +138 -0
  10. unicode_logic_kit/ace/runner.py +551 -0
  11. unicode_logic_kit/ace/translate.py +452 -0
  12. unicode_logic_kit/ace/verbalize.py +1070 -0
  13. unicode_logic_kit/api.py +1284 -0
  14. unicode_logic_kit/atp/__init__.py +177 -0
  15. unicode_logic_kit/atp/_ascii_names.py +113 -0
  16. unicode_logic_kit/atp/_html.py +72 -0
  17. unicode_logic_kit/atp/_substructural_input.py +228 -0
  18. unicode_logic_kit/atp/_tff_problem.py +715 -0
  19. unicode_logic_kit/atp/_tptp_problem.py +1111 -0
  20. unicode_logic_kit/atp/_writer_support.py +289 -0
  21. unicode_logic_kit/atp/clingo_backend.py +1180 -0
  22. unicode_logic_kit/atp/cvc5_backend.py +1385 -0
  23. unicode_logic_kit/atp/eprover_backend.py +732 -0
  24. unicode_logic_kit/atp/finite_domain.py +1055 -0
  25. unicode_logic_kit/atp/fitch.py +1547 -0
  26. unicode_logic_kit/atp/fitch_search.py +551 -0
  27. unicode_logic_kit/atp/hets_backend.py +339 -0
  28. unicode_logic_kit/atp/hybrid_down.py +120 -0
  29. unicode_logic_kit/atp/incremental.py +250 -0
  30. unicode_logic_kit/atp/kripke_enum.py +741 -0
  31. unicode_logic_kit/atp/lambek.py +436 -0
  32. unicode_logic_kit/atp/leo3_backend.py +332 -0
  33. unicode_logic_kit/atp/linear.py +738 -0
  34. unicode_logic_kit/atp/lj.py +705 -0
  35. unicode_logic_kit/atp/logic_backends.py +566 -0
  36. unicode_logic_kit/atp/ltl_tableau.py +1084 -0
  37. unicode_logic_kit/atp/minizinc_backend.py +1402 -0
  38. unicode_logic_kit/atp/modal_tableau.py +1382 -0
  39. unicode_logic_kit/atp/nanocop_backend.py +410 -0
  40. unicode_logic_kit/atp/portfolio.py +489 -0
  41. unicode_logic_kit/atp/protocol.py +1803 -0
  42. unicode_logic_kit/atp/prover9_entailment.py +1153 -0
  43. unicode_logic_kit/atp/resolution.py +1376 -0
  44. unicode_logic_kit/atp/resolution_check.py +1114 -0
  45. unicode_logic_kit/atp/sequent.py +1050 -0
  46. unicode_logic_kit/atp/tableau.py +921 -0
  47. unicode_logic_kit/atp/tableau_check.py +543 -0
  48. unicode_logic_kit/atp/tptp_ncl.py +811 -0
  49. unicode_logic_kit/atp/tptp_tff.py +1546 -0
  50. unicode_logic_kit/atp/tstp.py +1333 -0
  51. unicode_logic_kit/atp/tstp_check.py +1096 -0
  52. unicode_logic_kit/atp/twee_backend.py +236 -0
  53. unicode_logic_kit/atp/twee_check.py +711 -0
  54. unicode_logic_kit/atp/twee_entailment.py +953 -0
  55. unicode_logic_kit/atp/vampire_entailment.py +540 -0
  56. unicode_logic_kit/atp/z3_arith.py +470 -0
  57. unicode_logic_kit/atp/z3_equivalence.py +36 -0
  58. unicode_logic_kit/atp/z3_fuzzy.py +362 -0
  59. unicode_logic_kit/atp/z3_input.py +500 -0
  60. unicode_logic_kit/atp/z3_models.py +208 -0
  61. unicode_logic_kit/chem/__init__.py +88 -0
  62. unicode_logic_kit/chem/_naming.py +284 -0
  63. unicode_logic_kit/chem/cache.py +185 -0
  64. unicode_logic_kit/chem/interop.py +244 -0
  65. unicode_logic_kit/chem/mol.py +525 -0
  66. unicode_logic_kit/chem/signature.py +112 -0
  67. unicode_logic_kit/comorphism.py +497 -0
  68. unicode_logic_kit/dl/__init__.py +384 -0
  69. unicode_logic_kit/dl/classification.py +227 -0
  70. unicode_logic_kit/dl/concepts.py +632 -0
  71. unicode_logic_kit/dl/datatypes.py +818 -0
  72. unicode_logic_kit/dl/owl_functional.py +2433 -0
  73. unicode_logic_kit/dl/owl_manchester.py +1637 -0
  74. unicode_logic_kit/dl/owl_reasoner.py +790 -0
  75. unicode_logic_kit/dl/parser.py +391 -0
  76. unicode_logic_kit/dl/tableau.py +4048 -0
  77. unicode_logic_kit/dl/translate.py +2704 -0
  78. unicode_logic_kit/drt/__init__.py +94 -0
  79. unicode_logic_kit/drt/export.py +179 -0
  80. unicode_logic_kit/drt/nodes.py +506 -0
  81. unicode_logic_kit/drt/parser.py +965 -0
  82. unicode_logic_kit/drt/resolve.py +195 -0
  83. unicode_logic_kit/drt/reverse.py +175 -0
  84. unicode_logic_kit/eval/__init__.py +106 -0
  85. unicode_logic_kit/eval/batch.py +382 -0
  86. unicode_logic_kit/eval/canonical.py +663 -0
  87. unicode_logic_kit/eval/chem_batch.py +606 -0
  88. unicode_logic_kit/eval/converses.py +200 -0
  89. unicode_logic_kit/eval/datasets/__init__.py +136 -0
  90. unicode_logic_kit/eval/datasets/_base.py +263 -0
  91. unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
  92. unicode_logic_kit/eval/datasets/c3po.py +678 -0
  93. unicode_logic_kit/eval/datasets/folio.py +158 -0
  94. unicode_logic_kit/eval/datasets/fracas.py +418 -0
  95. unicode_logic_kit/eval/datasets/groves.py +191 -0
  96. unicode_logic_kit/eval/datasets/logicbench.py +467 -0
  97. unicode_logic_kit/eval/datasets/logicnli.py +303 -0
  98. unicode_logic_kit/eval/datasets/malls.py +133 -0
  99. unicode_logic_kit/eval/datasets/pfolio.py +594 -0
  100. unicode_logic_kit/eval/datasets/pmb.py +242 -0
  101. unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
  102. unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
  103. unicode_logic_kit/eval/datasets/proverqa.py +674 -0
  104. unicode_logic_kit/eval/datasets/willow.py +478 -0
  105. unicode_logic_kit/eval/equivalence.py +466 -0
  106. unicode_logic_kit/eval/exercise_gen.py +533 -0
  107. unicode_logic_kit/eval/explain.py +791 -0
  108. unicode_logic_kit/eval/generality.py +750 -0
  109. unicode_logic_kit/eval/metric_hf.py +458 -0
  110. unicode_logic_kit/eval/predicate_match.py +343 -0
  111. unicode_logic_kit/eval/theory_check.py +1170 -0
  112. unicode_logic_kit/eval/validate.py +306 -0
  113. unicode_logic_kit/fol/__init__.py +177 -0
  114. unicode_logic_kit/fol/_atom_keys.py +510 -0
  115. unicode_logic_kit/fol/_fol_nodes.py +3586 -0
  116. unicode_logic_kit/fol/_free_parameters.py +105 -0
  117. unicode_logic_kit/fol/_ho_nodes.py +448 -0
  118. unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
  119. unicode_logic_kit/fol/_identifiers.py +1091 -0
  120. unicode_logic_kit/fol/_lambek_nodes.py +112 -0
  121. unicode_logic_kit/fol/_linear_nodes.py +352 -0
  122. unicode_logic_kit/fol/_modal_nodes.py +1467 -0
  123. unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
  124. unicode_logic_kit/fol/_numeral_symbols.py +231 -0
  125. unicode_logic_kit/fol/_so_nodes.py +200 -0
  126. unicode_logic_kit/fol/_symbol_names.py +81 -0
  127. unicode_logic_kit/fol/_team_nodes.py +181 -0
  128. unicode_logic_kit/fol/_tptp_symbols.py +551 -0
  129. unicode_logic_kit/fol/_truth_constants.py +117 -0
  130. unicode_logic_kit/fol/casl_export.py +1135 -0
  131. unicode_logic_kit/fol/casl_import.py +929 -0
  132. unicode_logic_kit/fol/derivation.py +367 -0
  133. unicode_logic_kit/fol/dialect_detect.py +70 -0
  134. unicode_logic_kit/fol/dialect_repair.py +537 -0
  135. unicode_logic_kit/fol/frames.py +637 -0
  136. unicode_logic_kit/fol/grammars/terminals.lark +31 -0
  137. unicode_logic_kit/fol/lambda_tools.py +297 -0
  138. unicode_logic_kit/fol/latex_input.py +429 -0
  139. unicode_logic_kit/fol/modal_translation.py +944 -0
  140. unicode_logic_kit/fol/msflparser.py +1033 -0
  141. unicode_logic_kit/fol/naming.py +422 -0
  142. unicode_logic_kit/fol/nodes.py +241 -0
  143. unicode_logic_kit/fol/normalforms.py +492 -0
  144. unicode_logic_kit/fol/pal.py +287 -0
  145. unicode_logic_kit/fol/prolog_export.py +566 -0
  146. unicode_logic_kit/fol/prolog_input.py +505 -0
  147. unicode_logic_kit/fol/prover9_input.py +1325 -0
  148. unicode_logic_kit/fol/qml.py +1760 -0
  149. unicode_logic_kit/fol/qmltp_input.py +525 -0
  150. unicode_logic_kit/fol/sanitize.py +221 -0
  151. unicode_logic_kit/fol/serialize.py +79 -0
  152. unicode_logic_kit/fol/signature.py +1290 -0
  153. unicode_logic_kit/fol/simplify_check.py +544 -0
  154. unicode_logic_kit/fol/spans.py +594 -0
  155. unicode_logic_kit/fol/tptp_input.py +1503 -0
  156. unicode_logic_kit/fol/tptp_repair.py +941 -0
  157. unicode_logic_kit/fol/unification.py +157 -0
  158. unicode_logic_kit/fol/verbalize.py +263 -0
  159. unicode_logic_kit/hets/__init__.py +163 -0
  160. unicode_logic_kit/hets/bridge.py +142 -0
  161. unicode_logic_kit/hets/client.py +748 -0
  162. unicode_logic_kit/hets/docker.py +420 -0
  163. unicode_logic_kit/hets/dol.py +712 -0
  164. unicode_logic_kit/hets/haskell_json.py +355 -0
  165. unicode_logic_kit/hets/owl_backend.py +794 -0
  166. unicode_logic_kit/hets/owl_cli.py +598 -0
  167. unicode_logic_kit/hets/symbols.py +512 -0
  168. unicode_logic_kit/hol/__init__.py +140 -0
  169. unicode_logic_kit/hol/_ho_common.py +323 -0
  170. unicode_logic_kit/hol/_isabelle_binders.py +125 -0
  171. unicode_logic_kit/hol/classical.py +812 -0
  172. unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
  173. unicode_logic_kit/hol/deepshallow/_common.py +177 -0
  174. unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
  175. unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
  176. unicode_logic_kit/hol/deepshallow/modal.py +217 -0
  177. unicode_logic_kit/hol/deepshallow/qml.py +406 -0
  178. unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
  179. unicode_logic_kit/hol/free.py +753 -0
  180. unicode_logic_kit/hol/goedel.py +336 -0
  181. unicode_logic_kit/hol/ho_modal.py +1743 -0
  182. unicode_logic_kit/hol/intuitionistic.py +403 -0
  183. unicode_logic_kit/hol/isabelle_conditional.py +593 -0
  184. unicode_logic_kit/hol/isabelle_modal.py +1908 -0
  185. unicode_logic_kit/hol/isabelle_relevant.py +412 -0
  186. unicode_logic_kit/hol/isabelle_runner.py +1147 -0
  187. unicode_logic_kit/hol/isabelle_substructural.py +884 -0
  188. unicode_logic_kit/hol/lean.py +1018 -0
  189. unicode_logic_kit/hol/manyvalued.py +921 -0
  190. unicode_logic_kit/hol/secondorder.py +687 -0
  191. unicode_logic_kit/hol/thf_modal.py +941 -0
  192. unicode_logic_kit/hol/thirdorder.py +397 -0
  193. unicode_logic_kit/ilp/__init__.py +89 -0
  194. unicode_logic_kit/ilp/readback.py +389 -0
  195. unicode_logic_kit/ilp/separation.py +153 -0
  196. unicode_logic_kit/ilp/task.py +730 -0
  197. unicode_logic_kit/logic.py +163 -0
  198. unicode_logic_kit/mcp/__init__.py +28 -0
  199. unicode_logic_kit/mcp/__main__.py +5 -0
  200. unicode_logic_kit/mcp/chem_tools.py +1031 -0
  201. unicode_logic_kit/mcp/server.py +2453 -0
  202. unicode_logic_kit/mcp/syntax_spec.py +681 -0
  203. unicode_logic_kit/prob/__init__.py +53 -0
  204. unicode_logic_kit/prob/_bdd.py +225 -0
  205. unicode_logic_kit/prob/_column_gen.py +668 -0
  206. unicode_logic_kit/prob/distribution.py +686 -0
  207. unicode_logic_kit/prob/nilsson.py +470 -0
  208. unicode_logic_kit/py.typed +0 -0
  209. unicode_logic_kit/semantics/__init__.py +137 -0
  210. unicode_logic_kit/semantics/_modal_reject.py +156 -0
  211. unicode_logic_kit/semantics/action_models.py +466 -0
  212. unicode_logic_kit/semantics/asp_models.py +1200 -0
  213. unicode_logic_kit/semantics/conditional.py +580 -0
  214. unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
  215. unicode_logic_kit/semantics/free_logic.py +913 -0
  216. unicode_logic_kit/semantics/fuzzy.py +384 -0
  217. unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
  218. unicode_logic_kit/semantics/intuitionistic.py +581 -0
  219. unicode_logic_kit/semantics/kripke.py +1139 -0
  220. unicode_logic_kit/semantics/manyvalued.py +580 -0
  221. unicode_logic_kit/semantics/matrix.py +342 -0
  222. unicode_logic_kit/semantics/model_eval.py +1135 -0
  223. unicode_logic_kit/semantics/modelfinder.py +1036 -0
  224. unicode_logic_kit/semantics/nonmonotonic.py +372 -0
  225. unicode_logic_kit/semantics/relevant.py +331 -0
  226. unicode_logic_kit/semantics/secondorder.py +657 -0
  227. unicode_logic_kit/semantics/structures.py +352 -0
  228. unicode_logic_kit/semantics/tarski.py +975 -0
  229. unicode_logic_kit/semantics/team.py +315 -0
  230. unicode_logic_kit/semantics/team_translation.py +416 -0
  231. unicode_logic_kit/semantics/thirdorder.py +358 -0
  232. unicode_logic_kit/semantics/tnorm.py +85 -0
  233. unicode_logic_kit/semantics/truthtable.py +201 -0
  234. unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
  235. unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
  236. unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
  237. unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,391 @@
1
+ """A string parser for ALC(Q) concepts, dual to :meth:`Concept.to_unicode`.
2
+
3
+ ``dl.concepts`` builds concepts only by Python construction (``dl.And(A,
4
+ dl.Not(B))``); this module parses the glyph syntax that
5
+ :meth:`~unicode_logic_kit.dl.concepts.Concept.to_unicode` emits back into a
6
+ :class:`~unicode_logic_kit.dl.concepts.Concept`, so a rendered concept — or one
7
+ typed by hand in the same notation — round-trips: ``parse_concept(c.to_unicode())
8
+ == c`` for every constructor, including the qualified number restrictions
9
+ ``AtLeast``/``AtMost`` (≥n r.C / ≤n r.C), with the documented exceptions below.
10
+
11
+ The reader folds a flat chain of one connective to the LEFT (``A ⊓ B ⊓ C`` is
12
+ ``(A ⊓ B) ⊓ C``), so the printer writes a nested operand of the SAME connective
13
+ without parentheses on the left only: ``And(A, And(B, C))`` is ``A ⊓ (B ⊓ C)``,
14
+ never ``A ⊓ B ⊓ C``, which reads back as the other tree. The round trip is
15
+ therefore exact for every shape, not only for left-nested chains.
16
+
17
+ The first exceptions are the two constructors that name an INDIVIDUAL:
18
+ :class:`~unicode_logic_kit.dl.concepts.Nominal` renders as ``{a}`` and
19
+ :class:`~unicode_logic_kit.dl.concepts.HasValue` as ``∃r.{a}``, and neither reads
20
+ back — they are refused BY NAME instead (see ``_primary``). The glyph syntax
21
+ has no individual-name layer, so ``{a}`` cannot be told apart from a concept
22
+ NAME, and ``∃r.{a}`` cannot be told apart from ``∃r.`` applied to a nominal:
23
+ ``HasValue("r", "a")`` and ``Exists("r", Nominal("a"))`` have the SAME
24
+ rendering, honestly, because they have the same models — but picking one of
25
+ them when reading the text back would be a silent normalisation of the other.
26
+ This is the same render-only asymmetry :func:`~unicode_logic_kit.dl.to_manchester`
27
+ already has for a nominal, and the two syntaxes that DO distinguish the pair,
28
+ Manchester (``r value a`` versus ``r some {a}``) and Functional
29
+ (``ObjectHasValue(r a)`` versus ``ObjectSomeValuesFrom(r ObjectOneOf(a))``),
30
+ round-trip it exactly.
31
+
32
+ The DATA restrictions (:class:`~unicode_logic_kit.dl.concepts.DataExists` and its
33
+ four siblings) are render-only in the same way, for the same reason: the glyph
34
+ syntax has no data-range layer, and ``∃d.xsd:integer`` is the text of BOTH
35
+ ``DataExists("d", Datatype("xsd:integer"))`` and ``Exists("d",
36
+ Atomic("xsd:integer"))`` -- two concepts about different sorts. Reading it as
37
+ the second would be the silent misreading ``dl.parse_manchester`` had for
38
+ ``d some xsd:integer`` before the data layer. So a NAME that is a BUILT-IN
39
+ datatype (``xsd:integer``, ``rdfs:Literal``, …) is refused by name below; OWL 2
40
+ forbids a class with a datatype's name anyway. A USER-defined datatype
41
+ (``∃d.Digit``) cannot be told from a class by its spelling and reads as the
42
+ object restriction -- the limit the Manchester reader avoids by being told the
43
+ datatype names (``datatypes=``). Read data restrictions with
44
+ ``dl.parse_manchester`` or ``dl.parse_owl_functional_class_expression``.
45
+
46
+ An :class:`~unicode_logic_kit.dl.concepts.InverseRole` is the last one: it prints
47
+ as ``r⁻`` (``∃r⁻.C``), and the reader refuses a role name that ends in ``⁻`` BY
48
+ NAME (see ``_role_name``) instead of reading a plain role called ``r⁻``, which
49
+ would turn an inverse role into an unrelated one without a word.
50
+
51
+ Names. The grammar below says which names the reader can read. A name outside
52
+ it (one that holds whitespace, an ``.`` or a parenthesis, one of the reserved
53
+ glyphs, or nothing at all) is printed as it is, for display, and the reader then
54
+ refuses the text: this syntax has no escape (the Manchester and Functional-Style
55
+ writers bracket such a name as a full IRI). The one thing the printer does not do
56
+ is print a name so that the text reads back as ANOTHER concept — ``Atomic("A ⊓ B")``
57
+ would print ``A ⊓ B``, the intersection of two classes — and
58
+ ``Concept.to_unicode`` raises :class:`ValueError` for it.
59
+
60
+ Grammar (loosest-binding first, matching ``concepts.py``'s ``_PREC`` table
61
+ exactly — ⊔ at precedence 1, ⊓ at 2, ¬/∃/∀/≥/≤ at 3, atoms/⊤/⊥ at 4)::
62
+
63
+ concept := or
64
+ or := and ("⊔" and)*
65
+ and := unary ("⊓" unary)*
66
+ unary := "¬" unary
67
+ | ("∃" | "∀") NAME "." unary
68
+ | ("≥" | "≤") NUMBER NAME "." unary
69
+ | primary
70
+ primary := "⊤" | "⊥" | NAME | "(" concept ")"
71
+ NAME := a maximal run of characters that are none of: whitespace,
72
+ the glyphs ⊤ ⊥ ¬ ⊓ ⊔ ∃ ∀ ≥ ≤ ( ) . ⊑ { }
73
+ NUMBER := a NAME token consisting only of ASCII digits
74
+
75
+ A concept/role NAME may be any length and contain any characters outside
76
+ that reserved set (digits, underscores, non-ASCII letters, …) — e.g.
77
+ ``hasChild``, ``Doctor42``, ``θ``. ``{`` and ``}`` are in that reserved set
78
+ since 0.30.0 and are refused BY NAME (see ``_primary``): a nominal-shaped
79
+ text like ``{a}`` was a legal NAME before, so ``parse_concept("{a}")``
80
+ returned ``Atomic("{a}")`` -- a concept with a bogus class name -- silently,
81
+ for text ``dl.concepts`` itself prints. A concept name containing a literal
82
+ brace therefore stops parsing; there is no such name in the kit, its tests or
83
+ the OEO ontology this was measured against.
84
+ Restricting ``unary``'s operand (rather
85
+ than the full ``concept``) is what makes ``∃r.∀s.C`` and ``¬¬C`` parse
86
+ without parentheses while ``∃r.(C ⊓ D)`` requires them, mirroring
87
+ ``concepts.py``'s ``_paren`` exactly — so the grammar is precedence-faithful
88
+ by construction, not just by testing. ``≥``/``≤`` need a NUMBER token (the
89
+ bound ``n``) between the glyph and the role name, rendered by
90
+ ``Concept.to_unicode`` with a mandatory space before the role name (``"≥2
91
+ r.C"``, never ``"≥2r.C"``) so the tokenizer — which has no notion of a
92
+ digit/letter boundary — can always split the two apart; see ``_unary`` below.
93
+
94
+ This module intentionally implements a small hand-rolled recursive-descent
95
+ parser rather than reusing the Lark-based registry machinery in
96
+ ``fol/msflparser.py``: that machinery is purpose-built for assembling several
97
+ large, mutually-exclusive FOL dialects (FOL/MSFOL/MSFL/modal/…) that share a
98
+ term/atom/lambda layer, none of which applies here — ALC concepts are a tiny,
99
+ fixed, single-purpose grammar with no term layer, no dialects, and no
100
+ sharing to gain from the registry. A ~150-line hand-rolled parser is both
101
+ simpler and easier to audit for this shape of grammar than standing up a
102
+ Lark grammar file plus transformer for it.
103
+
104
+ Glyph-only: there is no ASCII fallback syntax (no ``E`` for ``∃``, ``A`` for
105
+ ``∀``, ``&`` for ``⊓``, …). Unlike the operator glyphs, ``A`` and ``E`` are
106
+ exactly the kind of short identifiers real concept/role names use (as they
107
+ do throughout ``tests/test_dl_alc.py``), so ASCII keyword fallbacks for the
108
+ quantifier-like operators would silently swallow a large fraction of
109
+ plausible concept names — not "trivially cheap" — so they are not provided.
110
+
111
+ GCIs (general concept inclusions, ``C ⊑ D``): ``concepts.py`` has no
112
+ renderer for them (:class:`~unicode_logic_kit.dl.tableau.TBox` carries
113
+ ``(Concept, Concept)`` pairs with no ``to_unicode``), so there is no
114
+ round-trip counterpart to test :func:`parse_gci` against — it is provided as
115
+ a convenience for hand-written GCI text using ⊑, the glyph the ``TBox`` /
116
+ ``subsumes`` docstrings already use for this notion, and is verified directly
117
+ against hand-picked expected ``(Concept, Concept)`` pairs instead.
118
+ """
119
+
120
+ from typing import List, Tuple
121
+
122
+ from .datatypes import is_builtin_datatype
123
+ from .concepts import (
124
+ Concept, Top, Bottom, Atomic, Not, And, Or, Exists, ForAll, AtLeast, AtMost,
125
+ )
126
+ from .tableau import RoleExpressionError, _reject_concept_role
127
+
128
+ __all__ = ["parse_concept", "parse_gci", "ConceptSyntaxError"]
129
+
130
+
131
+ class ConceptSyntaxError(ValueError):
132
+ """Raised by :func:`parse_concept` / :func:`parse_gci` on malformed input."""
133
+
134
+
135
+ # --------------------------------------------------------------------------- #
136
+ # Tokenizer.
137
+ # --------------------------------------------------------------------------- #
138
+
139
+ _GLYPH_TOKENS = {
140
+ "⊤": "TOP", "⊥": "BOT", "¬": "NOT", "⊓": "AND", "⊔": "OR",
141
+ "∃": "EXISTS", "∀": "FORALL", "≥": "ATLEAST", "≤": "ATMOST",
142
+ "(": "LPAREN", ")": "RPAREN", ".": "DOT", "⊑": "SUBSUME",
143
+ # RESERVED since 0.30.0, so `{a}` is refused BY NAME instead of read as a
144
+ # concept NAME spelled "{a}" -- see _primary's LBRACE branch.
145
+ "{": "LBRACE", "}": "RBRACE",
146
+ }
147
+
148
+ # A Token is (type: str, value: str, pos: int).
149
+ _Token = Tuple[str, str, int]
150
+
151
+
152
+ def _tokenize(text: str) -> List[_Token]:
153
+ """Split ``text`` into glyph tokens and maximal-run NAME tokens, plus a
154
+ trailing EOF sentinel (whose ``pos`` is ``len(text)``, for error messages).
155
+ """
156
+ tokens: List[_Token] = []
157
+ i, n = 0, len(text)
158
+ while i < n:
159
+ ch = text[i]
160
+ if ch.isspace():
161
+ i += 1
162
+ continue
163
+ if ch in _GLYPH_TOKENS:
164
+ tokens.append((_GLYPH_TOKENS[ch], ch, i))
165
+ i += 1
166
+ continue
167
+ start = i
168
+ while i < n and not text[i].isspace() and text[i] not in _GLYPH_TOKENS:
169
+ i += 1
170
+ tokens.append(("NAME", text[start:i], start))
171
+ tokens.append(("EOF", "", n))
172
+ return tokens
173
+
174
+
175
+ # --------------------------------------------------------------------------- #
176
+ # Recursive-descent parser.
177
+ # --------------------------------------------------------------------------- #
178
+
179
+ class _Parser:
180
+ """A single parse of one token stream; not re-used across calls."""
181
+
182
+ def __init__(self, tokens: List[_Token], text: str):
183
+ self._tokens = tokens
184
+ self._text = text
185
+ self._i = 0
186
+
187
+ def _peek(self) -> _Token:
188
+ return self._tokens[self._i]
189
+
190
+ def _advance(self) -> _Token:
191
+ tok = self._tokens[self._i]
192
+ self._i += 1
193
+ return tok
194
+
195
+ def _error(self, message: str) -> ConceptSyntaxError:
196
+ return ConceptSyntaxError(f"{message} in {self._text!r}")
197
+
198
+ def _expect(self, ttype: str, what: str) -> _Token:
199
+ tok = self._peek()
200
+ if tok[0] != ttype:
201
+ found = "end of input" if tok[0] == "EOF" else f"{tok[1]!r}"
202
+ raise self._error(
203
+ f"parse_concept: expected {what} but found {found} "
204
+ f"at position {tok[2]}")
205
+ return self._advance()
206
+
207
+ def _expect_eof(self) -> None:
208
+ tok = self._peek()
209
+ if tok[0] != "EOF":
210
+ raise self._error(
211
+ f"parse_concept: unexpected trailing input {tok[1]!r} "
212
+ f"at position {tok[2]}")
213
+
214
+ # -- grammar levels, loosest first (mirrors concepts.py's _PREC) -------- #
215
+
216
+ def _or(self) -> Concept:
217
+ left = self._and()
218
+ while self._peek()[0] == "OR":
219
+ self._advance()
220
+ left = Or(left, self._and())
221
+ return left
222
+
223
+ def _and(self) -> Concept:
224
+ left = self._unary()
225
+ while self._peek()[0] == "AND":
226
+ self._advance()
227
+ left = And(left, self._unary())
228
+ return left
229
+
230
+ def _checked(self, concept: Concept, pos: int) -> Concept:
231
+ """``concept``, unless its role is an OWL 2 built-in property name (or
232
+ ``=``/``≠``) — refused BY NAME, the way ``dl.parse_owl_functional`` and
233
+ the Manchester parser refuse it. ``∃owl:topObjectProperty.A`` used to
234
+ read as an ordinary role of that name, whose verdict (satisfiable) is
235
+ not the universal property's (every element is related to itself)."""
236
+ try:
237
+ _reject_concept_role(concept, where="parse_concept")
238
+ except RoleExpressionError as exc:
239
+ raise self._error(f"{exc} (at position {pos})") from exc
240
+ return concept
241
+
242
+ def _unary(self) -> Concept:
243
+ ttype, _, pos = self._peek()
244
+ if ttype == "NOT":
245
+ self._advance()
246
+ return Not(self._unary())
247
+ if ttype in ("EXISTS", "FORALL"):
248
+ self._advance()
249
+ role = self._role_name()
250
+ self._expect("DOT", "'.' after the role name")
251
+ body = self._unary()
252
+ return self._checked(
253
+ Exists(role, body) if ttype == "EXISTS" else ForAll(role, body), pos)
254
+ if ttype in ("ATLEAST", "ATMOST"):
255
+ self._advance()
256
+ n = self._expect_number()
257
+ role = self._role_name()
258
+ self._expect("DOT", "'.' after the role name")
259
+ body = self._unary()
260
+ return self._checked(
261
+ AtLeast(n, role, body) if ttype == "ATLEAST" else AtMost(n, role, body),
262
+ pos)
263
+ return self._primary()
264
+
265
+ def _role_name(self) -> str:
266
+ """Consume the NAME token of a restriction's role.
267
+
268
+ A name that ends in the inverse glyph ``⁻`` is refused BY NAME: that is
269
+ how ``Concept.to_unicode`` writes ``Exists(InverseRole("r"), C)``
270
+ (``∃r⁻.C``), a role EXPRESSION this syntax has no layer for, so reading
271
+ the text as a role NAMED ``r⁻`` would silently turn an inverse role into
272
+ an unrelated plain one.
273
+ """
274
+ tok = self._expect("NAME", "a role name")
275
+ if tok[1].endswith("⁻"):
276
+ raise self._error(
277
+ f"parse_concept: {tok[1]!r} (position {tok[2]}) is the glyph "
278
+ f"spelling of an INVERSE role, which the glyph syntax cannot "
279
+ f"read — reading it as a role named {tok[1]!r} would silently "
280
+ f"change what it says. Build dl.InverseRole({tok[1][:-1]!r}) "
281
+ f"directly (the in-house tableau refuses it by name; the "
282
+ f"external reasoner, dl.owl_reasoner, decides it)")
283
+ return tok[1]
284
+
285
+ def _expect_number(self) -> int:
286
+ """Consume a NAME token of only ASCII digits (the ``n`` in ``≥n``/``≤n``)."""
287
+ tok = self._peek()
288
+ if tok[0] == "NAME" and tok[1].isdigit():
289
+ self._advance()
290
+ return int(tok[1])
291
+ found = "end of input" if tok[0] == "EOF" else f"{tok[1]!r}"
292
+ raise self._error(
293
+ f"parse_concept: expected a non-negative integer but found {found} "
294
+ f"at position {tok[2]}")
295
+
296
+ def _primary(self) -> Concept:
297
+ ttype, value, pos = self._peek()
298
+ if ttype == "TOP":
299
+ self._advance()
300
+ return Top()
301
+ if ttype == "BOT":
302
+ self._advance()
303
+ return Bottom()
304
+ if ttype in ("LBRACE", "RBRACE"):
305
+ # A nominal-shaped text. Until 0.30.0 braces were not reserved, so
306
+ # `{a}` was a legal NAME and parse_concept("{a}") returned
307
+ # Atomic("{a}") -- a concept with a bogus class name, silently, for
308
+ # text this module's own siblings print. dl.parse_manchester
309
+ # already refused the same text by name; now so does this.
310
+ #
311
+ # Teaching it to BUILD a nominal was the alternative and is wrong:
312
+ # `∃r.{a}` is AMBIGUOUS between dl.HasValue(r, "a") and
313
+ # dl.Exists(r, dl.Nominal("a")) -- the glyph syntax has no
314
+ # individual-name layer to tell them apart -- and always picking
315
+ # one would be a silent normalisation of the other.
316
+ raise self._error(
317
+ f"parse_concept: nominal and value concepts "
318
+ f"('{{a}}', '∃r.{{a}}') are not supported — the glyph syntax "
319
+ f"has no individual-name layer, so '{{a}}' cannot be told "
320
+ f"apart from a concept NAME, and '∃r.{{a}}' cannot be told "
321
+ f"apart from a value restriction (found {value!r} at position "
322
+ f"{pos}). Build dl.Nominal('a') or dl.HasValue(role, 'a') "
323
+ f"directly, or read 'r value a' with dl.parse_manchester / "
324
+ f"'ObjectHasValue(r a)' with "
325
+ f"dl.parse_owl_functional_class_expression")
326
+ if ttype == "NAME":
327
+ if is_builtin_datatype(value):
328
+ raise self._error(
329
+ f"parse_concept: {value!r} (position {pos}) is a built-in "
330
+ f"DATATYPE, not a class, and the glyph syntax has no "
331
+ f"data-range layer -- reading '∃d.{value}' as an object "
332
+ f"restriction would silently change what it says. Read a "
333
+ f"data restriction with dl.parse_manchester "
334
+ f"('d some {value}') or "
335
+ f"dl.parse_owl_functional_class_expression "
336
+ f"('DataSomeValuesFrom(d {value})')")
337
+ self._advance()
338
+ return Atomic(value)
339
+ if ttype == "LPAREN":
340
+ self._advance()
341
+ inner = self._or()
342
+ self._expect("RPAREN", "')'")
343
+ return inner
344
+ found = "end of input" if ttype == "EOF" else f"{value!r}"
345
+ raise self._error(
346
+ f"parse_concept: unexpected {found} at position {pos}; expected "
347
+ "'⊤', '⊥', a concept name, '¬', '∃', '∀', '≥', '≤', or '('")
348
+
349
+ # -- entry points --------------------------------------------------- #
350
+
351
+ def parse_concept(self) -> Concept:
352
+ c = self._or()
353
+ self._expect_eof()
354
+ return c
355
+
356
+ def parse_gci(self) -> Tuple[Concept, Concept]:
357
+ sub = self._or()
358
+ self._expect("SUBSUME", "'⊑'")
359
+ sup = self._or()
360
+ self._expect_eof()
361
+ return sub, sup
362
+
363
+
364
+ def parse_concept(text: str) -> Concept:
365
+ """Parse ``text`` (the ⊤ ⊥ ¬ ⊓ ⊔ ∃ ∀ ≥ ≤ glyph syntax) into a :class:`Concept`.
366
+
367
+ Round-trips against :meth:`Concept.to_unicode`: ``parse_concept(c.to_unicode())
368
+ == c`` for every concept ``c`` EXCEPT the ones the glyph syntax cannot read:
369
+ the two that name an individual
370
+ (:class:`~unicode_logic_kit.dl.concepts.Nominal` and
371
+ :class:`~unicode_logic_kit.dl.concepts.HasValue`, which render as ``{a}`` and
372
+ ``∃r.{a}`` and are refused by name on the way back in — see the module
373
+ docstring and :class:`~unicode_logic_kit.dl.concepts.HasValue`; the same
374
+ render-only asymmetry ``dl.to_manchester`` has for a nominal), the data
375
+ restrictions, and an inverse role (``∃r⁻.C``), all refused by name. The
376
+ shape of a chain is exact: ``And(A, And(B, C))`` is written ``A ⊓ (B ⊓ C)``.
377
+ Raises :class:`ConceptSyntaxError` on
378
+ malformed input (unbalanced parentheses, a missing '.' after a role name,
379
+ a stray operator, trailing garbage, …).
380
+ """
381
+ return _Parser(_tokenize(text), text).parse_concept()
382
+
383
+
384
+ def parse_gci(text: str) -> Tuple[Concept, Concept]:
385
+ """Parse a general concept inclusion ``"C ⊑ D"`` into ``(C, D)``.
386
+
387
+ See the module docstring for why this has no round-trip counterpart to
388
+ test against (``concepts.py`` never renders a GCI). Raises
389
+ :class:`ConceptSyntaxError` on malformed input.
390
+ """
391
+ return _Parser(_tokenize(text), text).parse_gci()