unicode-logic-kit 0.31.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. unicode_logic_kit/__init__.py +385 -0
  2. unicode_logic_kit/__main__.py +520 -0
  3. unicode_logic_kit/_deadline.py +219 -0
  4. unicode_logic_kit/ace/__init__.py +126 -0
  5. unicode_logic_kit/ace/_align.py +135 -0
  6. unicode_logic_kit/ace/chem_lexicon.py +128 -0
  7. unicode_logic_kit/ace/drs_reader.py +570 -0
  8. unicode_logic_kit/ace/mapping.py +666 -0
  9. unicode_logic_kit/ace/reverse_modal.py +138 -0
  10. unicode_logic_kit/ace/runner.py +551 -0
  11. unicode_logic_kit/ace/translate.py +452 -0
  12. unicode_logic_kit/ace/verbalize.py +1070 -0
  13. unicode_logic_kit/api.py +1284 -0
  14. unicode_logic_kit/atp/__init__.py +177 -0
  15. unicode_logic_kit/atp/_ascii_names.py +113 -0
  16. unicode_logic_kit/atp/_html.py +72 -0
  17. unicode_logic_kit/atp/_substructural_input.py +228 -0
  18. unicode_logic_kit/atp/_tff_problem.py +715 -0
  19. unicode_logic_kit/atp/_tptp_problem.py +1111 -0
  20. unicode_logic_kit/atp/_writer_support.py +289 -0
  21. unicode_logic_kit/atp/clingo_backend.py +1180 -0
  22. unicode_logic_kit/atp/cvc5_backend.py +1385 -0
  23. unicode_logic_kit/atp/eprover_backend.py +732 -0
  24. unicode_logic_kit/atp/finite_domain.py +1055 -0
  25. unicode_logic_kit/atp/fitch.py +1547 -0
  26. unicode_logic_kit/atp/fitch_search.py +551 -0
  27. unicode_logic_kit/atp/hets_backend.py +339 -0
  28. unicode_logic_kit/atp/hybrid_down.py +120 -0
  29. unicode_logic_kit/atp/incremental.py +250 -0
  30. unicode_logic_kit/atp/kripke_enum.py +741 -0
  31. unicode_logic_kit/atp/lambek.py +436 -0
  32. unicode_logic_kit/atp/leo3_backend.py +332 -0
  33. unicode_logic_kit/atp/linear.py +738 -0
  34. unicode_logic_kit/atp/lj.py +705 -0
  35. unicode_logic_kit/atp/logic_backends.py +566 -0
  36. unicode_logic_kit/atp/ltl_tableau.py +1084 -0
  37. unicode_logic_kit/atp/minizinc_backend.py +1402 -0
  38. unicode_logic_kit/atp/modal_tableau.py +1382 -0
  39. unicode_logic_kit/atp/nanocop_backend.py +410 -0
  40. unicode_logic_kit/atp/portfolio.py +489 -0
  41. unicode_logic_kit/atp/protocol.py +1803 -0
  42. unicode_logic_kit/atp/prover9_entailment.py +1153 -0
  43. unicode_logic_kit/atp/resolution.py +1376 -0
  44. unicode_logic_kit/atp/resolution_check.py +1114 -0
  45. unicode_logic_kit/atp/sequent.py +1050 -0
  46. unicode_logic_kit/atp/tableau.py +921 -0
  47. unicode_logic_kit/atp/tableau_check.py +543 -0
  48. unicode_logic_kit/atp/tptp_ncl.py +811 -0
  49. unicode_logic_kit/atp/tptp_tff.py +1546 -0
  50. unicode_logic_kit/atp/tstp.py +1333 -0
  51. unicode_logic_kit/atp/tstp_check.py +1096 -0
  52. unicode_logic_kit/atp/twee_backend.py +236 -0
  53. unicode_logic_kit/atp/twee_check.py +711 -0
  54. unicode_logic_kit/atp/twee_entailment.py +953 -0
  55. unicode_logic_kit/atp/vampire_entailment.py +540 -0
  56. unicode_logic_kit/atp/z3_arith.py +470 -0
  57. unicode_logic_kit/atp/z3_equivalence.py +36 -0
  58. unicode_logic_kit/atp/z3_fuzzy.py +362 -0
  59. unicode_logic_kit/atp/z3_input.py +500 -0
  60. unicode_logic_kit/atp/z3_models.py +208 -0
  61. unicode_logic_kit/chem/__init__.py +88 -0
  62. unicode_logic_kit/chem/_naming.py +284 -0
  63. unicode_logic_kit/chem/cache.py +185 -0
  64. unicode_logic_kit/chem/interop.py +244 -0
  65. unicode_logic_kit/chem/mol.py +525 -0
  66. unicode_logic_kit/chem/signature.py +112 -0
  67. unicode_logic_kit/comorphism.py +497 -0
  68. unicode_logic_kit/dl/__init__.py +384 -0
  69. unicode_logic_kit/dl/classification.py +227 -0
  70. unicode_logic_kit/dl/concepts.py +632 -0
  71. unicode_logic_kit/dl/datatypes.py +818 -0
  72. unicode_logic_kit/dl/owl_functional.py +2433 -0
  73. unicode_logic_kit/dl/owl_manchester.py +1637 -0
  74. unicode_logic_kit/dl/owl_reasoner.py +790 -0
  75. unicode_logic_kit/dl/parser.py +391 -0
  76. unicode_logic_kit/dl/tableau.py +4048 -0
  77. unicode_logic_kit/dl/translate.py +2704 -0
  78. unicode_logic_kit/drt/__init__.py +94 -0
  79. unicode_logic_kit/drt/export.py +179 -0
  80. unicode_logic_kit/drt/nodes.py +506 -0
  81. unicode_logic_kit/drt/parser.py +965 -0
  82. unicode_logic_kit/drt/resolve.py +195 -0
  83. unicode_logic_kit/drt/reverse.py +175 -0
  84. unicode_logic_kit/eval/__init__.py +106 -0
  85. unicode_logic_kit/eval/batch.py +382 -0
  86. unicode_logic_kit/eval/canonical.py +663 -0
  87. unicode_logic_kit/eval/chem_batch.py +606 -0
  88. unicode_logic_kit/eval/converses.py +200 -0
  89. unicode_logic_kit/eval/datasets/__init__.py +136 -0
  90. unicode_logic_kit/eval/datasets/_base.py +263 -0
  91. unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
  92. unicode_logic_kit/eval/datasets/c3po.py +678 -0
  93. unicode_logic_kit/eval/datasets/folio.py +158 -0
  94. unicode_logic_kit/eval/datasets/fracas.py +418 -0
  95. unicode_logic_kit/eval/datasets/groves.py +191 -0
  96. unicode_logic_kit/eval/datasets/logicbench.py +467 -0
  97. unicode_logic_kit/eval/datasets/logicnli.py +303 -0
  98. unicode_logic_kit/eval/datasets/malls.py +133 -0
  99. unicode_logic_kit/eval/datasets/pfolio.py +594 -0
  100. unicode_logic_kit/eval/datasets/pmb.py +242 -0
  101. unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
  102. unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
  103. unicode_logic_kit/eval/datasets/proverqa.py +674 -0
  104. unicode_logic_kit/eval/datasets/willow.py +478 -0
  105. unicode_logic_kit/eval/equivalence.py +466 -0
  106. unicode_logic_kit/eval/exercise_gen.py +533 -0
  107. unicode_logic_kit/eval/explain.py +791 -0
  108. unicode_logic_kit/eval/generality.py +750 -0
  109. unicode_logic_kit/eval/metric_hf.py +458 -0
  110. unicode_logic_kit/eval/predicate_match.py +343 -0
  111. unicode_logic_kit/eval/theory_check.py +1170 -0
  112. unicode_logic_kit/eval/validate.py +306 -0
  113. unicode_logic_kit/fol/__init__.py +177 -0
  114. unicode_logic_kit/fol/_atom_keys.py +510 -0
  115. unicode_logic_kit/fol/_fol_nodes.py +3586 -0
  116. unicode_logic_kit/fol/_free_parameters.py +105 -0
  117. unicode_logic_kit/fol/_ho_nodes.py +448 -0
  118. unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
  119. unicode_logic_kit/fol/_identifiers.py +1091 -0
  120. unicode_logic_kit/fol/_lambek_nodes.py +112 -0
  121. unicode_logic_kit/fol/_linear_nodes.py +352 -0
  122. unicode_logic_kit/fol/_modal_nodes.py +1467 -0
  123. unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
  124. unicode_logic_kit/fol/_numeral_symbols.py +231 -0
  125. unicode_logic_kit/fol/_so_nodes.py +200 -0
  126. unicode_logic_kit/fol/_symbol_names.py +81 -0
  127. unicode_logic_kit/fol/_team_nodes.py +181 -0
  128. unicode_logic_kit/fol/_tptp_symbols.py +551 -0
  129. unicode_logic_kit/fol/_truth_constants.py +117 -0
  130. unicode_logic_kit/fol/casl_export.py +1135 -0
  131. unicode_logic_kit/fol/casl_import.py +929 -0
  132. unicode_logic_kit/fol/derivation.py +367 -0
  133. unicode_logic_kit/fol/dialect_detect.py +70 -0
  134. unicode_logic_kit/fol/dialect_repair.py +537 -0
  135. unicode_logic_kit/fol/frames.py +637 -0
  136. unicode_logic_kit/fol/grammars/terminals.lark +31 -0
  137. unicode_logic_kit/fol/lambda_tools.py +297 -0
  138. unicode_logic_kit/fol/latex_input.py +429 -0
  139. unicode_logic_kit/fol/modal_translation.py +944 -0
  140. unicode_logic_kit/fol/msflparser.py +1033 -0
  141. unicode_logic_kit/fol/naming.py +422 -0
  142. unicode_logic_kit/fol/nodes.py +241 -0
  143. unicode_logic_kit/fol/normalforms.py +492 -0
  144. unicode_logic_kit/fol/pal.py +287 -0
  145. unicode_logic_kit/fol/prolog_export.py +566 -0
  146. unicode_logic_kit/fol/prolog_input.py +505 -0
  147. unicode_logic_kit/fol/prover9_input.py +1325 -0
  148. unicode_logic_kit/fol/qml.py +1760 -0
  149. unicode_logic_kit/fol/qmltp_input.py +525 -0
  150. unicode_logic_kit/fol/sanitize.py +221 -0
  151. unicode_logic_kit/fol/serialize.py +79 -0
  152. unicode_logic_kit/fol/signature.py +1290 -0
  153. unicode_logic_kit/fol/simplify_check.py +544 -0
  154. unicode_logic_kit/fol/spans.py +594 -0
  155. unicode_logic_kit/fol/tptp_input.py +1503 -0
  156. unicode_logic_kit/fol/tptp_repair.py +941 -0
  157. unicode_logic_kit/fol/unification.py +157 -0
  158. unicode_logic_kit/fol/verbalize.py +263 -0
  159. unicode_logic_kit/hets/__init__.py +163 -0
  160. unicode_logic_kit/hets/bridge.py +142 -0
  161. unicode_logic_kit/hets/client.py +748 -0
  162. unicode_logic_kit/hets/docker.py +420 -0
  163. unicode_logic_kit/hets/dol.py +712 -0
  164. unicode_logic_kit/hets/haskell_json.py +355 -0
  165. unicode_logic_kit/hets/owl_backend.py +794 -0
  166. unicode_logic_kit/hets/owl_cli.py +598 -0
  167. unicode_logic_kit/hets/symbols.py +512 -0
  168. unicode_logic_kit/hol/__init__.py +140 -0
  169. unicode_logic_kit/hol/_ho_common.py +323 -0
  170. unicode_logic_kit/hol/_isabelle_binders.py +125 -0
  171. unicode_logic_kit/hol/classical.py +812 -0
  172. unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
  173. unicode_logic_kit/hol/deepshallow/_common.py +177 -0
  174. unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
  175. unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
  176. unicode_logic_kit/hol/deepshallow/modal.py +217 -0
  177. unicode_logic_kit/hol/deepshallow/qml.py +406 -0
  178. unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
  179. unicode_logic_kit/hol/free.py +753 -0
  180. unicode_logic_kit/hol/goedel.py +336 -0
  181. unicode_logic_kit/hol/ho_modal.py +1743 -0
  182. unicode_logic_kit/hol/intuitionistic.py +403 -0
  183. unicode_logic_kit/hol/isabelle_conditional.py +593 -0
  184. unicode_logic_kit/hol/isabelle_modal.py +1908 -0
  185. unicode_logic_kit/hol/isabelle_relevant.py +412 -0
  186. unicode_logic_kit/hol/isabelle_runner.py +1147 -0
  187. unicode_logic_kit/hol/isabelle_substructural.py +884 -0
  188. unicode_logic_kit/hol/lean.py +1018 -0
  189. unicode_logic_kit/hol/manyvalued.py +921 -0
  190. unicode_logic_kit/hol/secondorder.py +687 -0
  191. unicode_logic_kit/hol/thf_modal.py +941 -0
  192. unicode_logic_kit/hol/thirdorder.py +397 -0
  193. unicode_logic_kit/ilp/__init__.py +89 -0
  194. unicode_logic_kit/ilp/readback.py +389 -0
  195. unicode_logic_kit/ilp/separation.py +153 -0
  196. unicode_logic_kit/ilp/task.py +730 -0
  197. unicode_logic_kit/logic.py +163 -0
  198. unicode_logic_kit/mcp/__init__.py +28 -0
  199. unicode_logic_kit/mcp/__main__.py +5 -0
  200. unicode_logic_kit/mcp/chem_tools.py +1031 -0
  201. unicode_logic_kit/mcp/server.py +2453 -0
  202. unicode_logic_kit/mcp/syntax_spec.py +681 -0
  203. unicode_logic_kit/prob/__init__.py +53 -0
  204. unicode_logic_kit/prob/_bdd.py +225 -0
  205. unicode_logic_kit/prob/_column_gen.py +668 -0
  206. unicode_logic_kit/prob/distribution.py +686 -0
  207. unicode_logic_kit/prob/nilsson.py +470 -0
  208. unicode_logic_kit/py.typed +0 -0
  209. unicode_logic_kit/semantics/__init__.py +137 -0
  210. unicode_logic_kit/semantics/_modal_reject.py +156 -0
  211. unicode_logic_kit/semantics/action_models.py +466 -0
  212. unicode_logic_kit/semantics/asp_models.py +1200 -0
  213. unicode_logic_kit/semantics/conditional.py +580 -0
  214. unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
  215. unicode_logic_kit/semantics/free_logic.py +913 -0
  216. unicode_logic_kit/semantics/fuzzy.py +384 -0
  217. unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
  218. unicode_logic_kit/semantics/intuitionistic.py +581 -0
  219. unicode_logic_kit/semantics/kripke.py +1139 -0
  220. unicode_logic_kit/semantics/manyvalued.py +580 -0
  221. unicode_logic_kit/semantics/matrix.py +342 -0
  222. unicode_logic_kit/semantics/model_eval.py +1135 -0
  223. unicode_logic_kit/semantics/modelfinder.py +1036 -0
  224. unicode_logic_kit/semantics/nonmonotonic.py +372 -0
  225. unicode_logic_kit/semantics/relevant.py +331 -0
  226. unicode_logic_kit/semantics/secondorder.py +657 -0
  227. unicode_logic_kit/semantics/structures.py +352 -0
  228. unicode_logic_kit/semantics/tarski.py +975 -0
  229. unicode_logic_kit/semantics/team.py +315 -0
  230. unicode_logic_kit/semantics/team_translation.py +416 -0
  231. unicode_logic_kit/semantics/thirdorder.py +358 -0
  232. unicode_logic_kit/semantics/tnorm.py +85 -0
  233. unicode_logic_kit/semantics/truthtable.py +201 -0
  234. unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
  235. unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
  236. unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
  237. unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,681 @@
1
+ """Machine-readable syntax specification, served over MCP.
2
+
3
+ An LLM writing formulas for this kit needs the rules it is about to break —
4
+ and it needs them AT THE MOMENT it breaks them, not buried in a system
5
+ prompt it saw ten thousand tokens ago. This module is the retrievable form
6
+ of the kit's surface syntax: naming conventions, operator precedence,
7
+ dialect selection, the counting quantifier, and the documented failure modes
8
+ of LLM-generated formulas, each as a small JSON payload the model can fetch
9
+ by topic when a parse fails.
10
+
11
+ **The spec cannot drift from the parser.** Every ``EXAMPLES`` entry names the
12
+ dialect it is written in and what it must render back as, and
13
+ ``tests/test_syntax_spec.py`` parses every single one and compares. A rule
14
+ here that the parser does not actually implement is therefore a failing test,
15
+ not a lie an agent discovers at 3am. Wherever a fact is already derivable
16
+ from live kit objects (the dialect signal table, the parser mode ladder), it
17
+ is READ from them rather than restated.
18
+
19
+ The intended loop is: generate → parse fails → the error names a topic →
20
+ ``syntax_spec(topic=...)`` → regenerate. That closes the correction loop
21
+ without inflating every prompt with the full grammar, which matters when the
22
+ generating model is a local vLLM deployment paying for every prompt token on
23
+ every one of thousands of classes.
24
+ """
25
+
26
+ from typing import Dict, List, Optional
27
+
28
+ __all__ = ["syntax_spec", "SPEC_TOPICS", "EXAMPLES", "DL_EXAMPLES"]
29
+
30
+ SPEC_TOPICS = (
31
+ "overview", "naming", "dialects", "operators", "quantifiers",
32
+ "counting", "chemistry", "description-logic", "errors",
33
+ )
34
+
35
+ #: (label, dialect, source text, expected unicode rendering). The single
36
+ #: source of truth for every example shown in any topic below — the test
37
+ #: module parses each one, so an example can never quietly go stale.
38
+ EXAMPLES = (
39
+ ("variable-vs-constant", "fol", "P(x) ∧ P(alice)", "P(x) ∧ P(alice)"),
40
+ ("function-term", "fol", "P(f(x))", "P(f(x))"),
41
+ ("equality", "fol", "alice = bob", "alice = bob"),
42
+ ("non-ascii-letter", "fol", "LostTo(x, świątek)", "LostTo(x, świątek)"),
43
+ ("underscore-continuation", "fol", "Sibling(dani_Shapiro, family_History)",
44
+ "Sibling(dani_Shapiro, family_History)"),
45
+ ("digit-leading-name", "fol", "Hosted(beijing, 2008SummerOlympics)",
46
+ "Hosted(beijing, 2008SummerOlympics)"),
47
+ ("caseless-script-is-term-valued", "fol", "P(中文)", "P(中文)"),
48
+ ("greek-is-a-constant-not-a-name", "fol", "P(α)", "P(α)"),
49
+ ("quoted-constant", "fol", "P('k2') ∧ Q('John Doe') ∧ R('Alice')",
50
+ "P('k2') ∧ Q('John Doe') ∧ R('Alice')"),
51
+ ("quoted-constant-with-an-escape", "fol", r"P('it\'s') ∧ Q('a\\b')",
52
+ r"P('it\'s') ∧ Q('a\\b')"),
53
+ ("quoted-constant-that-needs-no-quotes", "fol", "P('socrates')",
54
+ "P(socrates)"),
55
+ ("quoted-sorted-constant", "msfol", "P('k2':Mountain)", "P('k2':Mountain)"),
56
+ ("universal", "fol", "∀x (Dog(x) → Animal(x))", "∀x (Dog(x) → Animal(x))"),
57
+ ("existential", "fol", "∃x (Dog(x) ∧ Black(x))", "∃x (Dog(x) ∧ Black(x))"),
58
+ ("counting", "fol", "∃≥40 x Carbon(x)", "∃≥40 x Carbon(x)"),
59
+ ("counting-exact", "fol", "∃=2 x Ring(x)", "∃=2 x Ring(x)"),
60
+ ("sorted-quantifier", "msfol", "∀x:Person Mortal(x)", "∀x:Person Mortal(x)"),
61
+ ("tptp-annotated", "tptp", "fof(a1, axiom, ![X]: (p(X) => q(X))).",
62
+ "∀x (P(x) → Q(x))"),
63
+ ("tptp-bare", "tptp_bare", "![X]: (p(X) => q(X))", "∀x (P(x) → Q(x))"),
64
+ ("tptp-lowercase-predicates", "tptp_bare", "c(A1) & o(A2) & bDOUBLE(A1,A2)",
65
+ "C(a1) ∧ O(a2) ∧ BDOUBLE(a1, a2)"),
66
+ ("tptp-quoted-name", "tptp_bare", "'1,2-diacyl-sn-glycero-3-phosphocholine'",
67
+ "1,2-diacyl-sn-glycero-3-phosphocholine"),
68
+ ("tptp-biimplication-precedence", "tptp_bare", "p <=> a & b", "P ↔ A ∧ B"),
69
+ ("latex", "latex", r"\forall x (P(x) \rightarrow Q(x))", "∀x (P(x) → Q(x))"),
70
+ ("prover9", "prover9", "all X (P(X) -> Q(X))", "∀x (P(x) → Q(x))"),
71
+ ("modal", "modal", "□(P → ◇Q)", "□(P → ◇Q)"),
72
+ )
73
+
74
+
75
+ def _examples(*labels: str) -> List[dict]:
76
+ by_label = {label: (dialect, text, rendering)
77
+ for label, dialect, text, rendering in EXAMPLES}
78
+ out = []
79
+ for label in labels:
80
+ dialect, text, rendering = by_label[label]
81
+ out.append({"label": label, "dialect": dialect, "input": text,
82
+ "renders_as": rendering})
83
+ return out
84
+
85
+
86
+ #: (label, syntax, source text, expected unicode rendering) for the
87
+ #: description-logic topic's own examples — kept SEPARATE from
88
+ #: :data:`EXAMPLES` above because those are all self-checked (by whichever
89
+ #: test module owns that job) via ``api.parse_any``, which has no dialect for
90
+ #: ALC concept text or OWL Manchester Syntax (see
91
+ #: :mod:`unicode_logic_kit.dl.parser` / :mod:`unicode_logic_kit.dl.owl_manchester`
92
+ #: — two wholly separate, non-FOL grammars). Verifying these instead needs
93
+ #: ``dl.parse_concept``/``dl.parse_manchester`` directly (``syntax`` here
94
+ #: selects which one), the same "spec cannot drift from the parser"
95
+ #: guarantee via the CORRECT grammar rather than silently reusing the wrong
96
+ #: one — see ``tests/test_mcp_server.py``'s own description-logic-topic test
97
+ #: for that check.
98
+ DL_EXAMPLES = (
99
+ ("alc-and-exists", "alc", "Person ⊓ ∃hasChild.Doctor",
100
+ "Person ⊓ ∃hasChild.Doctor"),
101
+ ("alc-negated-conjunction", "alc", "¬(A ⊓ B)", "¬(A ⊓ B)"),
102
+ ("manchester-and-some", "manchester", "Person and hasChild some Doctor",
103
+ "Person ⊓ ∃hasChild.Doctor"),
104
+ ("manchester-only", "manchester", "hasChild only Doctor", "∀hasChild.Doctor"),
105
+ ("manchester-number-restriction", "manchester", "hasChild min 2 Person",
106
+ "≥2 hasChild.Person"),
107
+ ("manchester-thing-nothing", "manchester", "owl:Thing and not owl:Nothing",
108
+ "⊤ ⊓ ¬⊥"),
109
+ )
110
+
111
+
112
+ def _dl_examples(*labels: str) -> List[dict]:
113
+ by_label = {label: (syntax, text, rendering)
114
+ for label, syntax, text, rendering in DL_EXAMPLES}
115
+ out = []
116
+ for label in labels:
117
+ syntax, text, rendering = by_label[label]
118
+ out.append({"label": label, "syntax": syntax, "input": text,
119
+ "renders_as": rendering})
120
+ return out
121
+
122
+
123
+ def _overview() -> dict:
124
+ from ..fol.dialect_detect import DIALECT_SIGNALS
125
+ from ..api import _UNICODE_MODES
126
+
127
+ return {
128
+ "summary": (
129
+ "Formulas are passed as plain text. The dialect is auto-detected "
130
+ "unless you pass an explicit hint; detection nominates candidates "
131
+ "in a fixed order and always falls back to the kit's own unicode "
132
+ "surface syntax."),
133
+ "dialects_in_detection_order": (
134
+ [name for name, _signal, _ascii in DIALECT_SIGNALS] + ["unicode"]),
135
+ "unicode_modes": [name for name, _kwargs in _UNICODE_MODES],
136
+ "most_common_mistake": (
137
+ "Assuming a lowercase name is a predicate. In the kit's OWN "
138
+ "unicode syntax a single lowercase letter is a VARIABLE and a "
139
+ "predicate must start uppercase — see topic 'naming', which also "
140
+ "says how a constant of any such name is written, in single "
141
+ "quotes ('a', 'Alice', 'John Doe'). In TPTP the "
142
+ "convention is exactly inverted. If your vocabulary has lowercase "
143
+ "predicate names (as chemical signatures do), write TPTP."),
144
+ "topics": list(SPEC_TOPICS),
145
+ }
146
+
147
+
148
+ def _naming() -> dict:
149
+ return {
150
+ "summary": (
151
+ "The kit's unicode syntax decides a symbol's KIND from its shape, "
152
+ "with no declaration needed. Getting this wrong is the single "
153
+ "most common cause of a formula that parses into something other "
154
+ "than what was meant, or fails to parse at all. The alphabet is "
155
+ "not ASCII-only: any Unicode letter can open an identifier (Greek "
156
+ "is the one carved-out exception — see 'greek_is_reserved' "
157
+ "below), so every rule here is stated in terms of 'letter', never "
158
+ "'a-z' or 'A-Z'."),
159
+ "rules": [
160
+ {"kind": "classification",
161
+ "shape": "the identifier's FIRST character alone decides its "
162
+ "kind, and nothing else does: str.isupper() true on "
163
+ "that one character means predicate; anything else "
164
+ "means term-valued (variable/name/constant). This is "
165
+ "checked per-character on the running interpreter, so "
166
+ "it holds for every script, not just Latin.",
167
+ "matches": ["Dog(x) — 'D' is upper-signalling, so PREDICATE",
168
+ "dog — 'd' is not, so term-valued"],
169
+ "note": "A script with no upper/lower distinction at all — "
170
+ "Chinese, Arabic, Hebrew, Devanagari, and most others — "
171
+ "never satisfies str.isupper() for any of its letters, "
172
+ "so a bare identifier from such a script is ALWAYS "
173
+ "term-valued and can NEVER head an atom by itself. "
174
+ "There is no separate 'caseless' rule to memorise — "
175
+ "this is the ordinary classification rule applied "
176
+ "honestly to an alphabet with no case to signal with. "
177
+ "A predicate over a caseless-script domain still needs "
178
+ "an uppercase-first name from a script that HAS case."},
179
+ {"kind": "variable",
180
+ "shape": "one term-valued letter (any script, cased or "
181
+ "caseless), then optional trailing digits",
182
+ "matches": ["x", "y", "a", "x1", "a1", "ś", "中"],
183
+ "note": "'a1' is a VARIABLE, not a constant — digits do not "
184
+ "change the kind. This is unchanged from a plain-ASCII "
185
+ "reading of the rule; only the letter itself is no "
186
+ "longer restricted to 'a'-'z'. A CONSTANT with such a "
187
+ "name is written in single quotes ('a', 'a1') — see "
188
+ "'quoted_constant'."},
189
+ {"kind": "constant",
190
+ "shape": "term-valued, at least two letters (any script) with "
191
+ "one not first — OR one-or-more ASCII digits then a "
192
+ "letter, then more of the same",
193
+ "matches": ["alice", "rex", "socrates", "中文", "świątek",
194
+ "dani_Shapiro", "2008SummerOlympics"],
195
+ "note": "A digit-leading identifier ('2008SummerOlympics') is "
196
+ "ALWAYS term-valued, never a predicate — digits carry "
197
+ "no case to signal predicate-hood with. It is also "
198
+ "never confused with a number: '2008' alone still "
199
+ "lexes as NUMBER, only a trailing LETTER pulls a token "
200
+ "into this class. From the second character on, an "
201
+ "underscore is legal ('dani_Shapiro') — but never as "
202
+ "the FIRST character of any identifier ('_foo' is a "
203
+ "NamingError)."},
204
+ {"kind": "quoted_constant",
205
+ "shape": "a constant whose name is one letter (or one letter "
206
+ "and digits), starts upper-case, or holds a space or "
207
+ "punctuation is written in single quotes. The text "
208
+ "between the quotes is the name, exactly. Inside, a "
209
+ "quote is written \\' and a backslash \\\\; no other "
210
+ "escape exists, and '' (no name) is a syntax error.",
211
+ "matches": ["'a'", "'k2'", "'Alice'", "'G-910'", "'C++'",
212
+ "'John Doe'", "'1,2-diacyl'", "'it\\'s'"],
213
+ "note": "Without quotes k2 is a VARIABLE and Alice is no "
214
+ "constant, so a constant of such a name MUST be "
215
+ "quoted: 'k2' is the constant named k2. A name that "
216
+ "reads as a constant bare may be quoted too, and is the "
217
+ "same constant ('socrates' and socrates). A quoted "
218
+ "constant stands where a constant stands, as an "
219
+ "argument or an operand; in a many-sorted dialect "
220
+ "(msfol, msfl) a constant always carries its sort, "
221
+ "'k2':Mountain. A predicate, a function name, a "
222
+ "variable, a sort, an agent and a nominal have NO "
223
+ "quoted form: 'Foo'(x) is a syntax error. Close every "
224
+ "quote you open: P('x) ∧ Q(y') is read as the one "
225
+ "atom P of the single constant named x) ∧ Q(y. Write "
226
+ "such a formula in the Unicode syntax, not in LaTeX: "
227
+ "the LaTeX reader refuses a quote."},
228
+ {"kind": "predicate",
229
+ "shape": "one uppercase-signalling letter (str.isupper() true), "
230
+ "then letters/digits — never an underscore, never a "
231
+ "digit first",
232
+ "matches": ["P", "Dog", "PRED", "HasBond", "Ś"],
233
+ "note": "'p(x)' does NOT parse as a predicate application in "
234
+ "the unicode dialect, and neither does a caseless-"
235
+ "script identifier standing alone ('中文(x)' is a "
236
+ "Function term, not a complete formula) — see "
237
+ "'classification' above."},
238
+ {"kind": "function", "shape": "a term-valued name applied to arguments",
239
+ "matches": ["f(x)", "mother(alice)", "2008SummerOlympics(x)"],
240
+ "note": "A term-valued name followed by '(' is a function term; "
241
+ "the same name standing alone is a variable or "
242
+ "constant."},
243
+ ],
244
+ "greek_is_reserved": (
245
+ "Plain Unicode-letter widening does NOT extend to Greek and "
246
+ "Coptic (U+0370-U+03FF) or Greek Extended (U+1F00-U+1FFF): "
247
+ "'λ' opens a lambda term and 'μ' opens a measure term "
248
+ "regardless of position, and the plain lowercase Greek run "
249
+ "(αβγδεζηθικνξοπρστυφχψω) is a CONSTANT in its own right rather "
250
+ "than a letter eligible to start a multi-letter NAME or a "
251
+ "predicate — 'P(α)' parses α as one Constant, not as if it "
252
+ "were an ordinary term-valued letter. Greek UPPERCASE is not "
253
+ "eligible for predicate position either: it is excluded from "
254
+ "the widened classes entirely, not merely redirected, so 'Ω(x)' "
255
+ "is a NamingError, not a predicate application. If your "
256
+ "vocabulary needs a genuinely Greek-named predicate or "
257
+ "function, spell it in a different script."),
258
+ "tptp_convention_is_inverted": (
259
+ "In TPTP, lowercase is a predicate/function and UPPERCASE is a "
260
+ "variable: 'p(X)' means predicate p applied to variable X. On "
261
+ "import the kit normalises to its own convention, so TPTP 'c(A1)' "
262
+ "becomes 'C(a1)'. This is a faithful, injective renaming — but "
263
+ "expect the case to flip when you read a formula back."),
264
+ "examples": _examples("variable-vs-constant", "function-term",
265
+ "equality", "non-ascii-letter",
266
+ "underscore-continuation", "digit-leading-name",
267
+ "caseless-script-is-term-valued",
268
+ "greek-is-a-constant-not-a-name",
269
+ "quoted-constant",
270
+ "quoted-constant-with-an-escape",
271
+ "quoted-constant-that-needs-no-quotes",
272
+ "quoted-sorted-constant",
273
+ "tptp-lowercase-predicates"),
274
+ }
275
+
276
+
277
+ def _dialects() -> dict:
278
+ from ..fol.dialect_detect import DIALECT_SIGNALS
279
+
280
+ notes = {
281
+ "smtlib": "SMT-LIB 2 assertions; multiple asserts fold into their "
282
+ "conjunction.",
283
+ "tptp": "Annotated TPTP (fof/cnf/tff/thf). parse_formula handles ONE "
284
+ "formula; use a problem loader for whole files.",
285
+ "latex": "LaTeX math macros (\\forall, \\land, \\rightarrow, ...).",
286
+ "tptp_bare": "A bare TPTP formula without the fof(...) wrapper. "
287
+ "Nominated only for pure-ASCII input.",
288
+ "prover9": "Prover9/LADR syntax (all/exists, ->, <->, &, |). "
289
+ "Nominated only for pure-ASCII input.",
290
+ "unicode": "The kit's own surface syntax — the fallback, and the only "
291
+ "one covering every logic family the kit supports.",
292
+ }
293
+ return {
294
+ "summary": (
295
+ "Detection is signal-based and ordered; each candidate is tried "
296
+ "and a parse failure falls through to the next, so a wrong guess "
297
+ "costs nothing but the fallback is always the unicode ladder."),
298
+ "dialects": [
299
+ {"name": name, "ascii_only": ascii_only,
300
+ "signal_regex": signal.pattern, "note": notes.get(name, "")}
301
+ for name, signal, ascii_only in DIALECT_SIGNALS
302
+ ] + [{"name": "unicode", "ascii_only": False, "signal_regex": None,
303
+ "note": notes["unicode"]}],
304
+ "how_to_pin": (
305
+ "Pass dialect=<name> to any tool to skip detection. Use this "
306
+ "whenever you generate in a fixed target syntax — it turns a "
307
+ "silent mis-detection into an explicit parse error."),
308
+ "examples": _examples("tptp-annotated", "tptp-bare", "latex",
309
+ "prover9"),
310
+ }
311
+
312
+
313
+ def _operators() -> dict:
314
+ return {
315
+ "summary": (
316
+ "Precedence, loosest binding LAST. Unary negation binds tightest; "
317
+ "the biconditional loosest. Implication is right-associative. "
318
+ "READ 'unicode_grammar_is_stricter' BELOW before relying on the "
319
+ "numeric levels: they describe the ASCII dialects (TPTP, LaTeX, "
320
+ "Prover9); the kit's own unicode syntax refuses some of these "
321
+ "combinations outright instead of resolving them."),
322
+ "unicode_grammar_is_stricter": (
323
+ "In the kit's OWN unicode syntax, ∧, ∨ and ⊕ sit at ONE level and "
324
+ "may not be mixed without brackets: 'A ∧ B ∨ C' is a SYNTAX "
325
+ "ERROR, not a formula with a default precedence. A chain of the same "
326
+ "connective is fine ('A ∧ B ∧ C'), and the levels relative to → "
327
+ "and ↔ do apply ('A ∧ B → C' parses as '(A ∧ B) → C'). In TPTP "
328
+ "the same input resolves by precedence ('a & b | c' is "
329
+ "'(a & b) | c'). Bracket explicitly and the two agree."),
330
+ "table": [
331
+ {"operator": "negation", "unicode": "¬", "tptp": "~",
332
+ "latex": "\\lnot", "prover9": "-", "binds": "tightest"},
333
+ {"operator": "conjunction", "unicode": "∧", "tptp": "&",
334
+ "latex": "\\land", "prover9": "&",
335
+ "binds": "2 (ASCII dialects); one level with ∨/⊕ in unicode"},
336
+ {"operator": "disjunction", "unicode": "∨", "tptp": "|",
337
+ "latex": "\\lor", "prover9": "|",
338
+ "binds": "3 (ASCII dialects); one level with ∧/⊕ in unicode"},
339
+ {"operator": "exclusive or", "unicode": "⊕", "tptp": "<~>",
340
+ "latex": "\\oplus", "prover9": None,
341
+ "binds": "3 (ASCII dialects); one level with ∧/∨ in unicode"},
342
+ {"operator": "implication", "unicode": "→", "tptp": "=>",
343
+ "latex": "\\rightarrow", "prover9": "->",
344
+ "binds": "4, right-associative"},
345
+ {"operator": "biconditional", "unicode": "↔", "tptp": "<=>",
346
+ "latex": "\\leftrightarrow", "prover9": "<->", "binds": "loosest"},
347
+ ],
348
+ "bracketing_advice": (
349
+ "'p <=> a & b' means 'p <=> (a & b)' — conjunction binds tighter "
350
+ "than the biconditional, and this kit's parser reads it that way. "
351
+ "Write the brackets anyway: some other TPTP tools reject the "
352
+ "unbracketed form, and explicit brackets are never wrong."),
353
+ "examples": _examples("tptp-biimplication-precedence"),
354
+ }
355
+
356
+
357
+ def _quantifiers() -> dict:
358
+ return {
359
+ "summary": (
360
+ "A quantifier binds ONE variable and scopes over the following "
361
+ "prefix-level expression: an atom, a negation, another "
362
+ "quantifier, or a PARENTHESISED formula. To scope over a "
363
+ "connective, parenthesise it."),
364
+ "forms": [
365
+ {"meaning": "for all", "unicode": "∀x φ", "tptp": "![X]: φ",
366
+ "latex": "\\forall x\\, φ", "prover9": "all X φ"},
367
+ {"meaning": "there is", "unicode": "∃x φ", "tptp": "?[X]: φ",
368
+ "latex": "\\exists x\\, φ", "prover9": "exists X φ"},
369
+ {"meaning": "sorted (many-sorted mode only)",
370
+ "unicode": "∀x:Person φ", "tptp": None, "latex": None,
371
+ "prover9": None,
372
+ "note": "Requires the msfol dialect; plain 'fol' rejects the "
373
+ "':Sort' annotation."},
374
+ ],
375
+ "scope_pitfall": (
376
+ "'∀x P(x) → Q(x)' parses as '(∀x P(x)) → Q(x)' with a FREE x in "
377
+ "Q(x). Write '∀x (P(x) → Q(x))'. A free variable is reported by "
378
+ "the check tool — treat that report as a scoping bug, not a "
379
+ "cosmetic warning."),
380
+ "examples": _examples("universal", "existential", "sorted-quantifier"),
381
+ }
382
+
383
+
384
+ def _counting() -> dict:
385
+ return {
386
+ "summary": (
387
+ "The counting quantifier states a cardinality directly: '∃≥n x φ' "
388
+ "is true iff at least n DISTINCT individuals satisfy φ ('∃≤n' at "
389
+ "most, '∃=n' exactly). The bound stays symbolic, so a large n "
390
+ "costs nothing to write down."),
391
+ "forms": [
392
+ {"meaning": "at least n", "unicode": "∃≥n x φ"},
393
+ {"meaning": "at most n", "unicode": "∃≤n x φ"},
394
+ {"meaning": "exactly n", "unicode": "∃=n x φ"},
395
+ ],
396
+ "why_not_an_existential_chain": (
397
+ "Writing 'at least 40 carbons' as 40 nested existentials plus all "
398
+ "pairwise inequalities means 780 inequality literals for n=40 — a "
399
+ "documented cause of model-checking timeouts, and the model "
400
+ "checker treats separately introduced variables as distinct "
401
+ "anyway. '∃≥40 x Carbon(x)' is one node and is counted directly. "
402
+ "Prefer the counting quantifier for every threshold statement."),
403
+ "lowering": (
404
+ "For backends without native counting the kit expands to the "
405
+ "standard distinct-witnesses encoding automatically — you never "
406
+ "need to write that expansion by hand."),
407
+ "examples": _examples("counting", "counting-exact"),
408
+ }
409
+
410
+
411
+ def _chemistry() -> dict:
412
+ return {
413
+ "summary": (
414
+ "A molecule is a finite first-order structure: the non-hydrogen "
415
+ "atoms are the domain, atom properties are unary predicates, "
416
+ "bonds are binary predicates, and whole-molecule properties are "
417
+ "0-ary. A class definition is a sentence checked against that "
418
+ "structure."),
419
+ "critical_dialect_note": (
420
+ "This vocabulary uses LOWERCASE predicate names (c, o, n, s). In "
421
+ "the kit's unicode syntax those are variables/functions, so "
422
+ "chemical definitions MUST be written in TPTP, where lowercase is "
423
+ "a predicate. Pin it with dialect='tptp_bare' rather than relying "
424
+ "on detection."),
425
+ "signature": {
426
+ "atom_types": ["c", "o", "n", "s", "p", "h"],
427
+ "hydrogen_count": ["has_0_hs", "has_1_hs", "has_2_hs", "has_3_hs"],
428
+ "charge": ["charge0", "charge_m1", "charge_p1"],
429
+ "bonds_binary": ["bSINGLE", "bDOUBLE", "bTRIPLE", "bAROMATIC",
430
+ "bond", "has_bond_to"],
431
+ "molecule_global_nullary": ["net_charge_neutral",
432
+ "NetChargePositive",
433
+ "NetChargeNegative"],
434
+ "note": "Bond relations are stored SYMMETRICALLY (both "
435
+ "directions), so bSINGLE(a,b) implies bSINGLE(b,a) is "
436
+ "present as a separate pair — do not add symmetry axioms.",
437
+ },
438
+ "computed_predicates": (
439
+ "Ring membership, aromaticity and connectivity need transitive "
440
+ "closure and are therefore NOT first-order definable over this "
441
+ "signature. They are supplied as COMPUTED predicates on the "
442
+ "structure (in_ring, aromatic, same_fragment) and may be used "
443
+ "freely in a definition — the structure decides them "
444
+ "algorithmically."),
445
+ "class_definition_shape": (
446
+ "A class is defined as a 0-ary predicate equivalence: "
447
+ "'className <=> (superClass & ?[A1,A2]: (...))'. Do NOT give the "
448
+ "defined predicate an argument — 'className(X) <=> ...' leaves X "
449
+ "unbound on the right and is a documented failure mode."),
450
+ "examples": _examples("tptp-lowercase-predicates", "tptp-quoted-name"),
451
+ }
452
+
453
+
454
+ def _description_logic() -> dict:
455
+ return {
456
+ "summary": (
457
+ "unicode_logic_kit.dl reasons over the description logic ALCHQ "
458
+ "(ALC plus role hierarchies/transitive roles 'H'/'S' and "
459
+ "qualified number restrictions 'Q'). Its MCP tools (dl_* in "
460
+ "unicode_logic_kit.mcp.server) take CONCEPT text, never a full FOL "
461
+ "formula — parsed by one of two wholly separate grammars, chosen "
462
+ "by each tool's own `syntax` argument. This is a DIFFERENT input "
463
+ "language from every other topic in this spec: api.parse_any "
464
+ "cannot read either of them, and neither of them can read a FOL "
465
+ "formula."),
466
+ "syntaxes": [
467
+ {"syntax": "alc", "default": True,
468
+ "note": "The glyph syntax Concept.to_unicode() itself emits "
469
+ "(dl.parse_concept) — ⊤ ⊥ ¬ ⊓ ⊔ ∃ ∀ ≥ ≤, no ASCII "
470
+ "fallback for the operators (an ASCII 'A'/'E' keyword "
471
+ "would swallow real concept/role names, which commonly "
472
+ "look exactly like that)."},
473
+ {"syntax": "manchester", "default": False,
474
+ "note": "The W3C OWL 2 Manchester Syntax (dl.parse_manchester), "
475
+ "restricted to what ALCHQ can express — keyword-based "
476
+ "('and'/'or'/'not'/'some'/'only'/'min'/'max'/'exactly'), "
477
+ "the notation Protege and most OWL tooling show by "
478
+ "default."},
479
+ ],
480
+ "concept_constructors": [
481
+ {"meaning": "top (everything)", "alc": "⊤", "manchester": "owl:Thing"},
482
+ {"meaning": "bottom (nothing)", "alc": "⊥", "manchester": "owl:Nothing"},
483
+ {"meaning": "concept name", "alc": "Person", "manchester": "Person"},
484
+ {"meaning": "negation", "alc": "¬C", "manchester": "not C"},
485
+ {"meaning": "intersection", "alc": "C ⊓ D", "manchester": "C and D"},
486
+ {"meaning": "union", "alc": "C ⊔ D", "manchester": "C or D"},
487
+ {"meaning": "existential restriction", "alc": "∃r.C",
488
+ "manchester": "r some C"},
489
+ {"meaning": "value restriction", "alc": "∀r.C", "manchester": "r only C"},
490
+ {"meaning": "at-least number restriction", "alc": "≥n r.C",
491
+ "manchester": "r min n C (C optional, defaults to owl:Thing)"},
492
+ {"meaning": "at-most number restriction", "alc": "≤n r.C",
493
+ "manchester": "r max n C (C optional, defaults to owl:Thing)"},
494
+ {"meaning": "exact number restriction (desugars to ≥n ⊓ ≤n)",
495
+ "alc": None, "manchester": "r exactly n C (C optional)"},
496
+ ],
497
+ "precedence": (
498
+ "⊔ loosest, then ⊓, then ¬/∃/∀/≥/≤ (all equal), then atoms/⊤/⊥ "
499
+ "tightest — identical in both syntaxes (dl.parser and "
500
+ "dl.owl_manchester share the same lattice, just spelling the "
501
+ "operators as glyphs vs. keywords), so '∃r.C ⊓ D' and "
502
+ "'r some C and D' both mean '(∃r.C) ⊓ D', and a filler beyond a "
503
+ "bare name needs parentheses either way: '∃r.(C ⊓ D)' / "
504
+ "'r some (C and D)'."),
505
+ "rbox_axioms": [
506
+ {"meaning": "role inclusion r ⊑ s", "alc": None,
507
+ "manchester": "r SubPropertyOf s"},
508
+ {"meaning": "transitivity declaration Trans(r)", "alc": None,
509
+ "manchester": "r Characteristics: Transitive"},
510
+ ],
511
+ "tbox_gci_axioms": [
512
+ {"meaning": "concept inclusion C ⊑ D", "alc": None,
513
+ "manchester": "C SubClassOf D"},
514
+ {"meaning": "concept equivalence C ≡ D", "alc": None,
515
+ "manchester": "C EquivalentTo D"},
516
+ ],
517
+ "tool_json_shapes": {
518
+ "summary": (
519
+ "Every dl_* MCP tool takes concept/axiom TEXT under `syntax` "
520
+ "(as above) plus, where a TBox/ABox is needed, plain JSON "
521
+ "rows — never Python TBox/ABox objects directly."),
522
+ "tbox_rows": [
523
+ {"shape": "{\"sub\": <text>, \"sup\": <text>}",
524
+ "meaning": "general concept inclusion sub ⊑ sup"},
525
+ {"shape": "{\"equiv\": [<text>, <text>]}",
526
+ "meaning": "equivalence (added as the two inclusions)"},
527
+ {"shape": "{\"subrole\": <role>, \"suprole\": <role>}",
528
+ "meaning": "role inclusion subrole ⊑ suprole (RBox 'H')"},
529
+ {"shape": "{\"transitive\": <role>}",
530
+ "meaning": "Trans(role) declaration (RBox 'S')"},
531
+ ],
532
+ "abox_shape": {
533
+ "concepts": "[[individual, concept_text], ...] "
534
+ "(individual : concept)",
535
+ "roles": "[[a, b, role], ...] ((a, b) : role)",
536
+ "distinct": "[[a, b], ...] (a ≠ b — the ONLY thing that "
537
+ "forces two individuals apart; without a unique "
538
+ "name assumption, two names may otherwise denote "
539
+ "the same individual)",
540
+ },
541
+ },
542
+ "number_restrictions_need_simple_roles": (
543
+ "A qualified number restriction (≥n r.C / ≤n r.C, or Manchester "
544
+ "'min'/'max'/'exactly') may not target a role that is itself "
545
+ "transitive, or has a transitive sub-role via the RBox — "
546
+ "combining unrestricted transitivity with counting makes "
547
+ "satisfiability undecidable. Naming such a role raises "
548
+ "NonSimpleRoleError, a structured {\"error\": {...}} (not the "
549
+ "ok=False text-parse shape — the CONCEPT text parsed fine; it is "
550
+ "the REASONING step that refuses the combination)."),
551
+ "class_definition_note": (
552
+ "Unlike the 'chemistry' topic's FOL formulas, a DL concept has "
553
+ "no free variable to accidentally leave open — 'Person and "
554
+ "hasChild some Doctor' is already a closed description of a "
555
+ "SET of individuals, so there is no 0-ary-predicate convention "
556
+ "to remember here."),
557
+ "examples": _dl_examples(
558
+ "alc-and-exists", "alc-negated-conjunction",
559
+ "manchester-and-some", "manchester-only",
560
+ "manchester-number-restriction", "manchester-thing-nothing"),
561
+ }
562
+
563
+
564
+ def _errors() -> dict:
565
+ return {
566
+ "summary": (
567
+ "The failure modes observed in LLM-generated chemical FOL, with "
568
+ "the fix for each. The first three are measured failure classes "
569
+ "from a published evaluation, not hypotheticals."),
570
+ "classes": [
571
+ {"kind": "biimplication_bracketing",
572
+ "symptom": "'predicate(x) <=> condition1 & condition2' rejected "
573
+ "by some TPTP parsers.",
574
+ "fix": "Bracket the right-hand side: '<=> (condition1 & "
575
+ "condition2)'. This kit's parser accepts both and reads "
576
+ "them identically, and the repair tool can rewrite the "
577
+ "text for you.",
578
+ "topic": "operators"},
579
+ {"kind": "mixed_same_level_connectives",
580
+ "symptom": "'A ∧ B ∨ C' rejected with \"Unexpected character "
581
+ "'∨' … Hint: Cannot mix …\". The message names the "
582
+ "token in front of the connective, so it can read "
583
+ "like a complaint about that name — it is not.",
584
+ "fix": "Bracket the intended grouping: '(A ∧ B) ∨ C' or "
585
+ "'A ∧ (B ∨ C)'. The kit's unicode grammar puts ∧, ∨ and "
586
+ "⊕ on ONE level and refuses to guess between the two "
587
+ "readings; renaming anything will fail in the same place. "
588
+ "The ASCII/TPTP dialects do apply a precedence, which is "
589
+ "why the same input is accepted there.",
590
+ "topic": "operators"},
591
+ {"kind": "invalid_predicate_name",
592
+ "symptom": "A chemical name used as a predicate starts with a "
593
+ "digit or contains punctuation, e.g. '(2S)Flavan4One' "
594
+ "or '1,2-diacyl-sn-glycero-3-phosphocholine'.",
595
+ "fix": "What single quotes can hold depends on what the name "
596
+ "is. A CONSTANT of any name is written in single quotes "
597
+ "in the kit's own unicode syntax and in TPTP alike "
598
+ "('1,2-diacyl', 'John Doe'). A PREDICATE or a FUNCTION "
599
+ "name has no quoted form in the unicode syntax, where "
600
+ "'Foo'(x) is a syntax error; write the formula in TPTP "
601
+ "(dialect='tptp_bare'), where any characters are allowed "
602
+ "inside a quoted atomic word, or let repair_formula "
603
+ "rename the predicate (the original is kept in its "
604
+ "'names'). Do NOT invent a sanitised camel-case name by "
605
+ "hand: that silently discards chemically meaningful "
606
+ "prefixes and breaks the link to the ontology class.",
607
+ "topic": "naming"},
608
+ {"kind": "unbound_variable",
609
+ "symptom": "'threeOxoSteroid(X) <=> (steroid & ?[A1]: c(A1))' — "
610
+ "X appears on the left but is never bound on the "
611
+ "right.",
612
+ "fix": "A class definition is 0-ary: drop the argument, "
613
+ "'threeOxoSteroid <=> (...)'. Never 'fix' it by adding a "
614
+ "quantifier without knowing what was meant — that changes "
615
+ "the claim.",
616
+ "topic": "chemistry"},
617
+ {"kind": "redundant_inequalities",
618
+ "symptom": "Dozens of existential variables plus every pairwise "
619
+ "'!=' between them.",
620
+ "fix": "Use the counting quantifier. Separately introduced "
621
+ "existential variables are already treated as distinct "
622
+ "under the all_different convention, so the inequalities "
623
+ "are pure cost.",
624
+ "topic": "counting"},
625
+ {"kind": "over_general_definition",
626
+ "symptom": "The definition matches far more molecules than the "
627
+ "class has members (high recall, very low precision).",
628
+ "fix": "Check whether a tiny structure already satisfies it — a "
629
+ "definition of a 50-atom class satisfied by a 3-atom "
630
+ "model is under-constrained. Add the constraints that "
631
+ "distinguish the class from its superclass, not just the "
632
+ "ones it shares with it.",
633
+ "topic": "chemistry"},
634
+ ],
635
+ }
636
+
637
+
638
+ _TOPIC_BUILDERS = {
639
+ "overview": _overview, "naming": _naming, "dialects": _dialects,
640
+ "operators": _operators, "quantifiers": _quantifiers,
641
+ "counting": _counting, "chemistry": _chemistry,
642
+ "description-logic": _description_logic, "errors": _errors,
643
+ }
644
+
645
+
646
+ def syntax_spec(topic: str = "overview",
647
+ dialect: Optional[str] = None) -> Dict:
648
+ """Return the syntax specification for ``topic``.
649
+
650
+ Topics: ``overview`` (dialects and the one mistake everyone makes),
651
+ ``naming`` (what makes a symbol a variable, constant, predicate or
652
+ function — and how TPTP inverts it), ``dialects``, ``operators``
653
+ (precedence table), ``quantifiers`` (scope rules), ``counting`` (the
654
+ cardinality quantifier and why it replaces existential chains),
655
+ ``chemistry`` (molecule-as-structure signature), ``description-logic``
656
+ (ALCHQ concept syntax — glyph or OWL Manchester — plus the TBox/ABox JSON
657
+ row shapes the ``dl_*`` MCP tools expect), ``errors`` (measured LLM
658
+ failure modes with fixes). ``dialect`` narrows the examples to one
659
+ FOL-family dialect, or (``description-logic`` only) one DL ``syntax``
660
+ (``"alc"``/``"manchester"``), where that makes sense.
661
+
662
+ Raises:
663
+ ValueError: unknown ``topic`` — the message lists the valid ones.
664
+ """
665
+ if topic not in _TOPIC_BUILDERS:
666
+ raise ValueError(
667
+ f"syntax_spec: unknown topic {topic!r} (one of "
668
+ f"{list(SPEC_TOPICS)})")
669
+ spec = dict(_TOPIC_BUILDERS[topic]())
670
+ spec["topic"] = topic
671
+ if dialect is not None and "examples" in spec:
672
+ # description-logic's own EXAMPLES-alike (DL_EXAMPLES) keys its
673
+ # per-example dialect as "syntax" (alc/manchester), never "dialect"
674
+ # (see DL_EXAMPLES's own docstring for why it is a separate list) —
675
+ # every other topic's examples use "dialect", so this falls back to
676
+ # "syntax" only for that one topic rather than requiring every
677
+ # example dict in the module to carry both keys.
678
+ key = "dialect" if topic != "description-logic" else "syntax"
679
+ spec["examples"] = [e for e in spec["examples"] if e[key] == dialect]
680
+ spec["filtered_to_dialect"] = dialect
681
+ return spec