unicode-logic-kit 0.31.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. unicode_logic_kit/__init__.py +385 -0
  2. unicode_logic_kit/__main__.py +520 -0
  3. unicode_logic_kit/_deadline.py +219 -0
  4. unicode_logic_kit/ace/__init__.py +126 -0
  5. unicode_logic_kit/ace/_align.py +135 -0
  6. unicode_logic_kit/ace/chem_lexicon.py +128 -0
  7. unicode_logic_kit/ace/drs_reader.py +570 -0
  8. unicode_logic_kit/ace/mapping.py +666 -0
  9. unicode_logic_kit/ace/reverse_modal.py +138 -0
  10. unicode_logic_kit/ace/runner.py +551 -0
  11. unicode_logic_kit/ace/translate.py +452 -0
  12. unicode_logic_kit/ace/verbalize.py +1070 -0
  13. unicode_logic_kit/api.py +1284 -0
  14. unicode_logic_kit/atp/__init__.py +177 -0
  15. unicode_logic_kit/atp/_ascii_names.py +113 -0
  16. unicode_logic_kit/atp/_html.py +72 -0
  17. unicode_logic_kit/atp/_substructural_input.py +228 -0
  18. unicode_logic_kit/atp/_tff_problem.py +715 -0
  19. unicode_logic_kit/atp/_tptp_problem.py +1111 -0
  20. unicode_logic_kit/atp/_writer_support.py +289 -0
  21. unicode_logic_kit/atp/clingo_backend.py +1180 -0
  22. unicode_logic_kit/atp/cvc5_backend.py +1385 -0
  23. unicode_logic_kit/atp/eprover_backend.py +732 -0
  24. unicode_logic_kit/atp/finite_domain.py +1055 -0
  25. unicode_logic_kit/atp/fitch.py +1547 -0
  26. unicode_logic_kit/atp/fitch_search.py +551 -0
  27. unicode_logic_kit/atp/hets_backend.py +339 -0
  28. unicode_logic_kit/atp/hybrid_down.py +120 -0
  29. unicode_logic_kit/atp/incremental.py +250 -0
  30. unicode_logic_kit/atp/kripke_enum.py +741 -0
  31. unicode_logic_kit/atp/lambek.py +436 -0
  32. unicode_logic_kit/atp/leo3_backend.py +332 -0
  33. unicode_logic_kit/atp/linear.py +738 -0
  34. unicode_logic_kit/atp/lj.py +705 -0
  35. unicode_logic_kit/atp/logic_backends.py +566 -0
  36. unicode_logic_kit/atp/ltl_tableau.py +1084 -0
  37. unicode_logic_kit/atp/minizinc_backend.py +1402 -0
  38. unicode_logic_kit/atp/modal_tableau.py +1382 -0
  39. unicode_logic_kit/atp/nanocop_backend.py +410 -0
  40. unicode_logic_kit/atp/portfolio.py +489 -0
  41. unicode_logic_kit/atp/protocol.py +1803 -0
  42. unicode_logic_kit/atp/prover9_entailment.py +1153 -0
  43. unicode_logic_kit/atp/resolution.py +1376 -0
  44. unicode_logic_kit/atp/resolution_check.py +1114 -0
  45. unicode_logic_kit/atp/sequent.py +1050 -0
  46. unicode_logic_kit/atp/tableau.py +921 -0
  47. unicode_logic_kit/atp/tableau_check.py +543 -0
  48. unicode_logic_kit/atp/tptp_ncl.py +811 -0
  49. unicode_logic_kit/atp/tptp_tff.py +1546 -0
  50. unicode_logic_kit/atp/tstp.py +1333 -0
  51. unicode_logic_kit/atp/tstp_check.py +1096 -0
  52. unicode_logic_kit/atp/twee_backend.py +236 -0
  53. unicode_logic_kit/atp/twee_check.py +711 -0
  54. unicode_logic_kit/atp/twee_entailment.py +953 -0
  55. unicode_logic_kit/atp/vampire_entailment.py +540 -0
  56. unicode_logic_kit/atp/z3_arith.py +470 -0
  57. unicode_logic_kit/atp/z3_equivalence.py +36 -0
  58. unicode_logic_kit/atp/z3_fuzzy.py +362 -0
  59. unicode_logic_kit/atp/z3_input.py +500 -0
  60. unicode_logic_kit/atp/z3_models.py +208 -0
  61. unicode_logic_kit/chem/__init__.py +88 -0
  62. unicode_logic_kit/chem/_naming.py +284 -0
  63. unicode_logic_kit/chem/cache.py +185 -0
  64. unicode_logic_kit/chem/interop.py +244 -0
  65. unicode_logic_kit/chem/mol.py +525 -0
  66. unicode_logic_kit/chem/signature.py +112 -0
  67. unicode_logic_kit/comorphism.py +497 -0
  68. unicode_logic_kit/dl/__init__.py +384 -0
  69. unicode_logic_kit/dl/classification.py +227 -0
  70. unicode_logic_kit/dl/concepts.py +632 -0
  71. unicode_logic_kit/dl/datatypes.py +818 -0
  72. unicode_logic_kit/dl/owl_functional.py +2433 -0
  73. unicode_logic_kit/dl/owl_manchester.py +1637 -0
  74. unicode_logic_kit/dl/owl_reasoner.py +790 -0
  75. unicode_logic_kit/dl/parser.py +391 -0
  76. unicode_logic_kit/dl/tableau.py +4048 -0
  77. unicode_logic_kit/dl/translate.py +2704 -0
  78. unicode_logic_kit/drt/__init__.py +94 -0
  79. unicode_logic_kit/drt/export.py +179 -0
  80. unicode_logic_kit/drt/nodes.py +506 -0
  81. unicode_logic_kit/drt/parser.py +965 -0
  82. unicode_logic_kit/drt/resolve.py +195 -0
  83. unicode_logic_kit/drt/reverse.py +175 -0
  84. unicode_logic_kit/eval/__init__.py +106 -0
  85. unicode_logic_kit/eval/batch.py +382 -0
  86. unicode_logic_kit/eval/canonical.py +663 -0
  87. unicode_logic_kit/eval/chem_batch.py +606 -0
  88. unicode_logic_kit/eval/converses.py +200 -0
  89. unicode_logic_kit/eval/datasets/__init__.py +136 -0
  90. unicode_logic_kit/eval/datasets/_base.py +263 -0
  91. unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
  92. unicode_logic_kit/eval/datasets/c3po.py +678 -0
  93. unicode_logic_kit/eval/datasets/folio.py +158 -0
  94. unicode_logic_kit/eval/datasets/fracas.py +418 -0
  95. unicode_logic_kit/eval/datasets/groves.py +191 -0
  96. unicode_logic_kit/eval/datasets/logicbench.py +467 -0
  97. unicode_logic_kit/eval/datasets/logicnli.py +303 -0
  98. unicode_logic_kit/eval/datasets/malls.py +133 -0
  99. unicode_logic_kit/eval/datasets/pfolio.py +594 -0
  100. unicode_logic_kit/eval/datasets/pmb.py +242 -0
  101. unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
  102. unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
  103. unicode_logic_kit/eval/datasets/proverqa.py +674 -0
  104. unicode_logic_kit/eval/datasets/willow.py +478 -0
  105. unicode_logic_kit/eval/equivalence.py +466 -0
  106. unicode_logic_kit/eval/exercise_gen.py +533 -0
  107. unicode_logic_kit/eval/explain.py +791 -0
  108. unicode_logic_kit/eval/generality.py +750 -0
  109. unicode_logic_kit/eval/metric_hf.py +458 -0
  110. unicode_logic_kit/eval/predicate_match.py +343 -0
  111. unicode_logic_kit/eval/theory_check.py +1170 -0
  112. unicode_logic_kit/eval/validate.py +306 -0
  113. unicode_logic_kit/fol/__init__.py +177 -0
  114. unicode_logic_kit/fol/_atom_keys.py +510 -0
  115. unicode_logic_kit/fol/_fol_nodes.py +3586 -0
  116. unicode_logic_kit/fol/_free_parameters.py +105 -0
  117. unicode_logic_kit/fol/_ho_nodes.py +448 -0
  118. unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
  119. unicode_logic_kit/fol/_identifiers.py +1091 -0
  120. unicode_logic_kit/fol/_lambek_nodes.py +112 -0
  121. unicode_logic_kit/fol/_linear_nodes.py +352 -0
  122. unicode_logic_kit/fol/_modal_nodes.py +1467 -0
  123. unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
  124. unicode_logic_kit/fol/_numeral_symbols.py +231 -0
  125. unicode_logic_kit/fol/_so_nodes.py +200 -0
  126. unicode_logic_kit/fol/_symbol_names.py +81 -0
  127. unicode_logic_kit/fol/_team_nodes.py +181 -0
  128. unicode_logic_kit/fol/_tptp_symbols.py +551 -0
  129. unicode_logic_kit/fol/_truth_constants.py +117 -0
  130. unicode_logic_kit/fol/casl_export.py +1135 -0
  131. unicode_logic_kit/fol/casl_import.py +929 -0
  132. unicode_logic_kit/fol/derivation.py +367 -0
  133. unicode_logic_kit/fol/dialect_detect.py +70 -0
  134. unicode_logic_kit/fol/dialect_repair.py +537 -0
  135. unicode_logic_kit/fol/frames.py +637 -0
  136. unicode_logic_kit/fol/grammars/terminals.lark +31 -0
  137. unicode_logic_kit/fol/lambda_tools.py +297 -0
  138. unicode_logic_kit/fol/latex_input.py +429 -0
  139. unicode_logic_kit/fol/modal_translation.py +944 -0
  140. unicode_logic_kit/fol/msflparser.py +1033 -0
  141. unicode_logic_kit/fol/naming.py +422 -0
  142. unicode_logic_kit/fol/nodes.py +241 -0
  143. unicode_logic_kit/fol/normalforms.py +492 -0
  144. unicode_logic_kit/fol/pal.py +287 -0
  145. unicode_logic_kit/fol/prolog_export.py +566 -0
  146. unicode_logic_kit/fol/prolog_input.py +505 -0
  147. unicode_logic_kit/fol/prover9_input.py +1325 -0
  148. unicode_logic_kit/fol/qml.py +1760 -0
  149. unicode_logic_kit/fol/qmltp_input.py +525 -0
  150. unicode_logic_kit/fol/sanitize.py +221 -0
  151. unicode_logic_kit/fol/serialize.py +79 -0
  152. unicode_logic_kit/fol/signature.py +1290 -0
  153. unicode_logic_kit/fol/simplify_check.py +544 -0
  154. unicode_logic_kit/fol/spans.py +594 -0
  155. unicode_logic_kit/fol/tptp_input.py +1503 -0
  156. unicode_logic_kit/fol/tptp_repair.py +941 -0
  157. unicode_logic_kit/fol/unification.py +157 -0
  158. unicode_logic_kit/fol/verbalize.py +263 -0
  159. unicode_logic_kit/hets/__init__.py +163 -0
  160. unicode_logic_kit/hets/bridge.py +142 -0
  161. unicode_logic_kit/hets/client.py +748 -0
  162. unicode_logic_kit/hets/docker.py +420 -0
  163. unicode_logic_kit/hets/dol.py +712 -0
  164. unicode_logic_kit/hets/haskell_json.py +355 -0
  165. unicode_logic_kit/hets/owl_backend.py +794 -0
  166. unicode_logic_kit/hets/owl_cli.py +598 -0
  167. unicode_logic_kit/hets/symbols.py +512 -0
  168. unicode_logic_kit/hol/__init__.py +140 -0
  169. unicode_logic_kit/hol/_ho_common.py +323 -0
  170. unicode_logic_kit/hol/_isabelle_binders.py +125 -0
  171. unicode_logic_kit/hol/classical.py +812 -0
  172. unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
  173. unicode_logic_kit/hol/deepshallow/_common.py +177 -0
  174. unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
  175. unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
  176. unicode_logic_kit/hol/deepshallow/modal.py +217 -0
  177. unicode_logic_kit/hol/deepshallow/qml.py +406 -0
  178. unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
  179. unicode_logic_kit/hol/free.py +753 -0
  180. unicode_logic_kit/hol/goedel.py +336 -0
  181. unicode_logic_kit/hol/ho_modal.py +1743 -0
  182. unicode_logic_kit/hol/intuitionistic.py +403 -0
  183. unicode_logic_kit/hol/isabelle_conditional.py +593 -0
  184. unicode_logic_kit/hol/isabelle_modal.py +1908 -0
  185. unicode_logic_kit/hol/isabelle_relevant.py +412 -0
  186. unicode_logic_kit/hol/isabelle_runner.py +1147 -0
  187. unicode_logic_kit/hol/isabelle_substructural.py +884 -0
  188. unicode_logic_kit/hol/lean.py +1018 -0
  189. unicode_logic_kit/hol/manyvalued.py +921 -0
  190. unicode_logic_kit/hol/secondorder.py +687 -0
  191. unicode_logic_kit/hol/thf_modal.py +941 -0
  192. unicode_logic_kit/hol/thirdorder.py +397 -0
  193. unicode_logic_kit/ilp/__init__.py +89 -0
  194. unicode_logic_kit/ilp/readback.py +389 -0
  195. unicode_logic_kit/ilp/separation.py +153 -0
  196. unicode_logic_kit/ilp/task.py +730 -0
  197. unicode_logic_kit/logic.py +163 -0
  198. unicode_logic_kit/mcp/__init__.py +28 -0
  199. unicode_logic_kit/mcp/__main__.py +5 -0
  200. unicode_logic_kit/mcp/chem_tools.py +1031 -0
  201. unicode_logic_kit/mcp/server.py +2453 -0
  202. unicode_logic_kit/mcp/syntax_spec.py +681 -0
  203. unicode_logic_kit/prob/__init__.py +53 -0
  204. unicode_logic_kit/prob/_bdd.py +225 -0
  205. unicode_logic_kit/prob/_column_gen.py +668 -0
  206. unicode_logic_kit/prob/distribution.py +686 -0
  207. unicode_logic_kit/prob/nilsson.py +470 -0
  208. unicode_logic_kit/py.typed +0 -0
  209. unicode_logic_kit/semantics/__init__.py +137 -0
  210. unicode_logic_kit/semantics/_modal_reject.py +156 -0
  211. unicode_logic_kit/semantics/action_models.py +466 -0
  212. unicode_logic_kit/semantics/asp_models.py +1200 -0
  213. unicode_logic_kit/semantics/conditional.py +580 -0
  214. unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
  215. unicode_logic_kit/semantics/free_logic.py +913 -0
  216. unicode_logic_kit/semantics/fuzzy.py +384 -0
  217. unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
  218. unicode_logic_kit/semantics/intuitionistic.py +581 -0
  219. unicode_logic_kit/semantics/kripke.py +1139 -0
  220. unicode_logic_kit/semantics/manyvalued.py +580 -0
  221. unicode_logic_kit/semantics/matrix.py +342 -0
  222. unicode_logic_kit/semantics/model_eval.py +1135 -0
  223. unicode_logic_kit/semantics/modelfinder.py +1036 -0
  224. unicode_logic_kit/semantics/nonmonotonic.py +372 -0
  225. unicode_logic_kit/semantics/relevant.py +331 -0
  226. unicode_logic_kit/semantics/secondorder.py +657 -0
  227. unicode_logic_kit/semantics/structures.py +352 -0
  228. unicode_logic_kit/semantics/tarski.py +975 -0
  229. unicode_logic_kit/semantics/team.py +315 -0
  230. unicode_logic_kit/semantics/team_translation.py +416 -0
  231. unicode_logic_kit/semantics/thirdorder.py +358 -0
  232. unicode_logic_kit/semantics/tnorm.py +85 -0
  233. unicode_logic_kit/semantics/truthtable.py +201 -0
  234. unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
  235. unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
  236. unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
  237. unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,965 @@
1
+ """Two parsers into :class:`~unicode_logic_kit.drt.nodes.DRS`: a compact hand-rolled box
2
+ notation (:func:`parse_drs`), and a documented SUBSET of the Parallel Meaning Bank's
3
+ Sequence Box Notation (:func:`parse_sbn`) — the PMB's line-based interchange format for
4
+ DRS-like structures (van Noord et al.'s Parallel Meaning Bank project releases sense-
5
+ disambiguated, role-annotated semantic parses in this format; it is the natural "found
6
+ data" source for DRSs, hence the SBN import path).
7
+
8
+ Both are small, purpose-built, single-shot grammars — no dialect sharing, no term/lambda
9
+ layer — so, exactly as ``unicode_logic_kit.dl.parser`` argues for ALC concepts, a hand-rolled
10
+ tokenizer/recursive-descent parser is simpler and gives better per-construct error messages
11
+ than standing up a Lark grammar + transformer for either of them.
12
+
13
+ =================================================================================
14
+ 1. The box notation (``parse_drs``)
15
+ =================================================================================
16
+
17
+ Grammar (``|`` = alternative, ``?`` = optional, ``*`` = zero-or-more; NAME classes are
18
+ exactly ``unicode_logic_kit.drt.nodes``'s ``is_referent`` / ``is_predicate_name`` /
19
+ ``is_constant_name``, i.e. the kit's own VARIABLE / PREDICATE / NAME lexical conventions)::
20
+
21
+ document := boxexpr EOF
22
+ boxexpr := "~" box # -> Neg(box)
23
+ | box (("->" | "∨") box)? # -> box | Impl(box, box) | Or(box, box)
24
+
25
+ box := "[" refs? "|" conds? "]"
26
+ refs := REF ("," REF)*
27
+ conds := cond ("," cond)*
28
+ cond := pred_cond | eq_cond
29
+ | "~" box # -> Neg(box)
30
+ | box ("->" | "∨") box # -> Impl(box, box) | Or(box, box)
31
+
32
+ pred_cond := PRED "(" term ("," term)* ")"
33
+ | "Card" "(" term "," CMP "," NUMBER ")" # -> Card (0.24.0; the operator
34
+ | # slot disambiguates against
35
+ | # a Pred merely NAMED Card)
36
+ | "Part_of" "(" term "," term ")" # -> Part (0.24.0; exactly 2 args)
37
+ eq_cond := term "=" term
38
+ term := REF | CONST | STRING
39
+ CMP := ">=" | "<=" | "=" | ">" | "<"
40
+ NUMBER := /[0-9]+/
41
+
42
+ REF : a single lowercase letter optionally followed by digits (x, y, e12, ...)
43
+ CONST : a lowercase-initial bare identifier of >= 2 characters (john, daisy, ...)
44
+ PRED : an uppercase-initial identifier, underscores allowed in
45
+ continuation (Farmer, Owns, Has_bond_to, ...)
46
+ STRING : a double-quoted literal ("John Doe") — sanitized to a legal CONST via
47
+ unicode_logic_kit.fol.sanitize.NameMapping (kept consistent within one
48
+ parse_drs call; not returned — see "Quoted constants" below)
49
+
50
+ Worked example (the classic donkey sentence, exactly as it appears in the kit roadmap)::
51
+
52
+ [x, y | Farmer(x), Donkey(y), Owns(x, y)] -> [ | Beats(x, y)]
53
+
54
+ parses to ``DRS((), (Impl(DRS(("x","y"), (Pred("Farmer",("x",)), Pred("Donkey",("y",)),
55
+ Pred("Owns",("x","y")))), DRS((), (Pred("Beats",("x","y")),))),))`` — a top-level bare
56
+ ``Impl``/``Neg``/``Or`` result is wrapped as the sole condition of an empty-referent DRS
57
+ (a "bare box" result, with no trailing ``->``/``∨``, IS the returned DRS directly).
58
+
59
+ **Deliberately unambiguous, at the cost of expressiveness.** ``boxexpr`` accepts AT MOST
60
+ ONE ``->``/``∨`` after a box, and ``~`` only ever prefixes a bare ``box`` (never a whole
61
+ ``boxexpr``) — so ``~[...] -> [...]`` and ``[...] -> [...] -> [...]`` are BOTH refused
62
+ (with a message naming the dangling trailing token) rather than silently picking a
63
+ left/right-associativity or an operator-precedence convention nobody asked for. Nest boxes
64
+ explicitly instead — e.g. ``[ | ~[...]] -> [...]`` for the first case — box notation has
65
+ no expressiveness loss, only a syntax-level one. A "cond" that is a bare box with no
66
+ ``->``/``∨`` suffix is refused too (naming the position): a bare DRS is not one of the five
67
+ condition variants (:mod:`unicode_logic_kit.drt.nodes`) — wrap it in ``~[...]`` or make it a
68
+ ``->``/``∨`` operand.
69
+
70
+ **Quoted constants.** A bare CONST token is already legal (``is_constant_name``) and passes
71
+ through unchanged. A quoted ``"..."`` STRING is for constants box notation's bare-token
72
+ syntax cannot express as-is (spaces, punctuation, digit-leading, single characters that
73
+ would otherwise collide with the referent namespace, non-ASCII, ...); it is sanitized via a
74
+ fresh :class:`~unicode_logic_kit.fol.sanitize.NameMapping` PER ``parse_drs`` CALL (consistent
75
+ within one call — the same quoted string always sanitizes to the same constant — but not
76
+ returned to the caller, since the primary, recommended path is to just write a legal bare
77
+ CONST directly, as the roadmap's own donkey-sentence facts do: ``Farmer(john)``, no quotes
78
+ needed). :func:`parse_sbn` below, whose quoted constants are the norm rather than an
79
+ escape hatch, DOES return its mapping.
80
+
81
+ =================================================================================
82
+ 2. The SBN subset (``parse_sbn``)
83
+ =================================================================================
84
+
85
+ Real SBN is a line-per-token format: each line names a WordNet-style sense for one token of
86
+ the sentence, optionally followed by semantic-role edges to OTHER lines (by relative line
87
+ offset) or to literal constants. This function implements a precisely bounded SUBSET,
88
+ documented here in full — anything outside it is REFUSED, naming the construct, rather than
89
+ silently mis-parsed::
90
+
91
+ sbn := line+
92
+ line := TABS (sense_line | "NEGATION") COMMENT?
93
+ sense_line := SENSE (ROLE target)*
94
+ SENSE := LEMMA "." POS "." SENSE_NUM # e.g. farmer.n.01
95
+ LEMMA : /[a-z][a-z_]*/ # underscore-joined lowercase words
96
+ POS : one of n, v, a, r, s
97
+ SENSE_NUM : exactly two digits
98
+ ROLE := /[A-Z][a-zA-Z0-9]*(-[A-Z][a-zA-Z0-9]*)?/ # e.g. Agent, Patient, Co-Theme
99
+ target := OFFSET | STRING
100
+ OFFSET : /[+-][0-9]+/ # relative CONTENT-line index (see below)
101
+ STRING : a double-quoted literal
102
+ COMMENT := "%" rest-of-line, ignored wherever it appears
103
+ blank (whitespace-only) lines are ignored and do not consume a content-line index.
104
+
105
+ * **Indentation is exactly TAB characters** (one tab = one nesting depth; mixing in spaces,
106
+ or indenting with spaces at all, is refused). Depth 0 lines form the outermost DRS; a
107
+ depth-``d`` line may only be followed by a depth-``d+1`` line if it is a ``NEGATION``
108
+ line (opening a sub-box that the following, deeper-indented lines belong to, up to but
109
+ not including the next line at depth <= d, or EOF) — any other depth jump (skipping a
110
+ level, or dedenting to a depth that was never open) is refused, naming the line.
111
+ * **Line numbering.** Every CONTENT line (a ``sense_line`` or a ``NEGATION`` line — blank
112
+ and comment-only lines do not count) gets a 1-based index in document order, REGARDLESS
113
+ of indentation depth; a ``sense_line`` at index ``i`` introduces the referent ``f"e{i}"``.
114
+ A ``NEGATION`` line consumes an index too (so offsets past it stay stable) but introduces
115
+ no referent of its own — targeting one is refused, naming the line ("a box-operator line
116
+ has no referent in this subset").
117
+ * **Sense -> predicate, exactly as the kit roadmap specifies**: ``person.n.01`` becomes the
118
+ predicate ``PersonN01`` (each dot-separated part capitalised — an underscore-joined lemma
119
+ like ``get_up.v.02`` becomes ``GetUpV02`` — and concatenated; PoS and sense-number keep
120
+ their literal digits/letter). The token -> predicate mapping is recorded and returned
121
+ (:class:`SBNMapping.predicates`) since it is lossy (case and the dots are gone).
122
+ * **Role -> binary predicate**, exactly as written (``Agent``) applied to
123
+ ``(this_line's_referent, target)`` — i.e. ``Agent(e3, e1)`` style, per the roadmap.
124
+ **A role name may carry one hyphenated segment** (``Co-Theme``, ``Co-Agent``,
125
+ ``Co-Patient``, ...) — VerbNet's own "Co-" compounding, common throughout real PMB
126
+ releases; the hyphen is dropped when the role becomes a predicate name (``Co-Theme``
127
+ -> ``CoTheme``, since the kit PREDICATE convention has no hyphen). This is part of the
128
+ BASE grammar above, in force in both SBN dialects this function reads (see 2b below) —
129
+ real PMB documents use ``Co-``-compounded roles whether or not they also happen to use
130
+ the connector dialect's own box-operator mechanism (e.g. a flat document with no
131
+ ``NEGATION`` line at all can still carry a ``Co-Theme`` role), so this widening cannot
132
+ be gated to one dialect without silently refusing real, in-scope input.
133
+ * **The only supported box-changing operator is ``NEGATION``.** Any other ALL-CAPS token
134
+ in sense-token position (``POSSIBLE``, ``NECESSARY``, ``DISCOURSE_REFERENCE``, PMB's
135
+ other real operators, ...) is REFUSED BY NAME: this subset is negation-only. A
136
+ ``NEGATION`` line with no more-indented line following it (an empty scope) is refused too
137
+ — in this subset a box operator only ever appears to introduce content, never vacuously.
138
+ * **Quoted constants** are sanitized via :class:`~unicode_logic_kit.fol.sanitize.NameMapping`
139
+ (lower-cased first, so ``"John"`` and ``"john"`` map together), and the mapping IS
140
+ returned (:class:`SBNMapping.constants`) — unlike the box notation's escape-hatch
141
+ quoting, a quoted literal is SBN's ONLY way to write a constant, so recovering the
142
+ original matters.
143
+ * **Bare (unquoted) constants**: the four deictic references Bos (2023) §2.1 documents
144
+ (``now``/``speaker``/``hearer``/``here`` — utterance time/speaker/addressee/location) and
145
+ a bare UNSIGNED integer (``Quantity 3``) are also accepted as constant targets, routed
146
+ through the same :class:`NameMapping` as quoted constants. Unambiguous with an OFFSET
147
+ target, which always carries a sign.
148
+ * **A comment may itself contain a ``%``, ANSI colour escapes, or start with the PMB release
149
+ format's own ``%%%``-prefixed generation-command header** — all of that is already inside
150
+ the comment by the ``COMMENT`` rule above (everything from the first ``%`` on the line),
151
+ so none of it needs special handling.
152
+
153
+ **What is explicitly out of scope** (refused by name, never silently dropped): event
154
+ quantification / plural referents, presupposition triggers, any operator besides NEGATION,
155
+ multi-word discourse (this parses ONE sbn "document" — i.e. one sentence's worth of boxes —
156
+ per call, matching this kit subpackage's single-sentence-plus-anaphora scope; see the
157
+ ``unicode_logic_kit.drt`` package docstring).
158
+
159
+ =================================================================================
160
+ 2b. A second SBN dialect: PMB's own released format (the "connector" dialect)
161
+ =================================================================================
162
+
163
+ The dialect above was hand-written before any real PMB release was measured against it.
164
+ PMB's actual gold releases (verified against pmb-5.1.0's
165
+ ``data/<lang>/gold/p<NN>/d<NNNN>/<lang>.drs.sbn`` files) use a DIFFERENT, but fully
166
+ documented, mechanism instead of textual indentation — Bos (2023), "The Sequence Notation:
167
+ Catching Complex Meanings in Simple Graphs" (IWCS 2023), §§2.1/3.4/4.1.
168
+ :func:`parse_sbn` recognizes it automatically (see "Dialect selection" below) and reads it
169
+ as follows — everything not listed here (LEMMA/POS/SENSE_NUM, roles, quoted constants,
170
+ comments, ...) is exactly as in the dialect above:
171
+
172
+ * **Leading whitespace is PURE COSMETIC COLUMN ALIGNMENT**, never nesting depth (verified:
173
+ no CONCEPT line in pmb-5.1.0 ever carries leading whitespace; only box-operator lines do,
174
+ padding them to the sense-token column for human readability) — this dialect ignores it
175
+ entirely, on every line, and never refuses a space the indentation dialect above would.
176
+ * **A box-operator line carries a trailing CONNECTOR token** (``<N``, a positive integer)
177
+ instead of relying on indentation. Contexts (boxes) are numbered by INTRODUCTION ORDER:
178
+ context 0 is the outermost, implicit context holding everything before the first
179
+ separator; the K-th separator encountered (1-based, document order) introduces context K.
180
+ Its connector ``<N`` (``1 <= N <= K``) says the separator's OWN reading (``Neg`` for
181
+ ``NEGATION``) is a CONDITION of context ``K - N`` — e.g. ``<1`` attaches to the immediately
182
+ preceding context, ``<2`` skips one further back, and so on; several separators may attach
183
+ to the SAME earlier context (e.g. "she is neither rich nor famous": ``NEGATION <1`` then
184
+ ``NEGATION <2``, both landing on context 0, giving ``¬Rich(x) ∧ ¬Famous(x)`` rather than a
185
+ nested double negation — Bos 2023 Figure 4). As above, only ``NEGATION`` is a supported
186
+ separator; every other name PMB emits (``POSSIBILITY``, ``NECESSITY``, and the SDRT
187
+ discourse relations ``CONTINUATION``, ``CONTRAST``, ``CONJUNCTION``, ...) is refused by
188
+ name — this subset's DRS conditions have no modal-box or discourse-relation reading for
189
+ them. A FORWARD connector (``>N``) is refused too (it needs two-pass resolution this
190
+ subset does not implement).
191
+ * **Role-hook indices (``+N``/``-N``) count CONCEPT (sense) lines ONLY**, skipping every
192
+ box-operator line entirely — DELIBERATELY DIFFERENT from the dialect above's own
193
+ numbering (which counts box-operator lines too, so that offsets stay stable across them):
194
+ this is what Bos (2023) §3.3 and pmb-5.1.0 itself actually implement. Changing the other
195
+ dialect's existing (self-consistent, if non-standard) numbering would break its own
196
+ already-accepted inputs, so both numbering conventions coexist, one per dialect.
197
+ * A role target that is itself a connector (``Proposition >1`` — an embedded-clause /
198
+ propositional-attitude argument pointing AT A CONTEXT rather than a concept) is refused by
199
+ name: this subset only resolves entity-valued role targets.
200
+ * Role names may carry one hyphenated segment (``Co-Theme``, ``Co-Agent``, ...) — this is
201
+ the BASE grammar's own rule (see the ``ROLE`` widening in section 2 above), not a
202
+ connector-dialect addition; it is repeated here only because ``Co-`` compounding is
203
+ especially common throughout pmb-5.1.0's connector-dialect documents. Comparison/
204
+ temporal/spatial OPERATOR tokens (``EQU``, ``TPR``, ...) are lexically indistinguishable
205
+ from roles in this subset and get the SAME uninterpreted binary-predicate treatment —
206
+ none of their axiomatic content (e.g. ``EQU``'s reflexivity) is asserted, only the atom
207
+ itself.
208
+
209
+ **Dialect selection.** :func:`parse_sbn` looks at every box-operator line up front: if ANY
210
+ of them carries a trailing connector token, the WHOLE document is read in the connector
211
+ dialect (every box-operator line must then carry one, or the specific line is named and
212
+ refused); otherwise it is read in the indentation dialect above (a bare ``NEGATION``, TAB
213
+ depth). The two dialects' numbering/indentation conventions are per-document, never mixed
214
+ within one call — this is exactly why every existing ``parse_sbn`` input (none of which
215
+ contains a connector token) keeps parsing to the identical DRS it always has.
216
+
217
+ The 2-3 SBN examples exercised in ``tests/test_drt.py`` for the indentation dialect are
218
+ CONSTRUCTED BY HAND — they are illustrative, not verbatim PMB corpus data. The connector
219
+ dialect is exercised against both hand-written fixtures and, when a caller points
220
+ ``UFK_PMB_SBN_DIR`` at one, a real PMB gold release — see
221
+ :mod:`unicode_logic_kit.eval.datasets.pmb` and ``tests/test_datasets_pmb.py``.
222
+ """
223
+
224
+ import re
225
+ from dataclasses import dataclass
226
+ from typing import Dict, List, Tuple, Union
227
+
228
+ from ..fol.sanitize import NameMapping
229
+ from .nodes import (
230
+ DRS, Card, Condition, Pred, Eq, Neg, Impl, Or, Part,
231
+ is_referent, is_predicate_name, is_constant_name,
232
+ )
233
+
234
+ __all__ = [
235
+ "parse_drs", "DRSSyntaxError",
236
+ "parse_sbn", "SBNSyntaxError", "SBNMapping",
237
+ ]
238
+
239
+
240
+ class DRSSyntaxError(ValueError):
241
+ """Raised by :func:`parse_drs` on malformed box-notation input."""
242
+
243
+
244
+ class SBNSyntaxError(ValueError):
245
+ """Raised by :func:`parse_sbn` on malformed input, or a construct outside the
246
+ documented SBN subset (see the module docstring)."""
247
+
248
+
249
+ # =============================================================================
250
+ # 1. Box notation.
251
+ # =============================================================================
252
+
253
+ _GLYPHS = {
254
+ "[": "LB", "]": "RB", "|": "PIPE", ",": "COMMA", "~": "TILDE",
255
+ "=": "EQ", "(": "LP", ")": "RP", "∨": "OR",
256
+ }
257
+
258
+ # A Token is (type: str, value: str, pos: int).
259
+ _Token = Tuple[str, str, int]
260
+
261
+
262
+ def _tokenize_drs(text: str) -> List[_Token]:
263
+ """Split ``text`` into box-notation tokens, plus a trailing EOF sentinel."""
264
+ tokens: List[_Token] = []
265
+ i, n = 0, len(text)
266
+ while i < n:
267
+ ch = text[i]
268
+ if ch.isspace():
269
+ i += 1
270
+ continue
271
+ if ch == "-" and i + 1 < n and text[i + 1] == ">":
272
+ tokens.append(("ARROW", "->", i))
273
+ i += 2
274
+ continue
275
+ if ch in "<>":
276
+ if i + 1 < n and text[i + 1] == "=":
277
+ tokens.append(("CMP", ch + "=", i))
278
+ i += 2
279
+ else:
280
+ tokens.append(("CMP", ch, i))
281
+ i += 1
282
+ continue
283
+ if ch.isdigit():
284
+ j = i
285
+ while j < n and text[j].isdigit():
286
+ j += 1
287
+ tokens.append(("NUMBER", text[i:j], i))
288
+ i = j
289
+ continue
290
+ if ch in _GLYPHS:
291
+ tokens.append((_GLYPHS[ch], ch, i))
292
+ i += 1
293
+ continue
294
+ if ch == '"':
295
+ j = i + 1
296
+ while j < n and text[j] != '"':
297
+ j += 1
298
+ if j >= n:
299
+ raise DRSSyntaxError(
300
+ f"parse_drs: unterminated string literal starting at position "
301
+ f"{i} in {text!r}")
302
+ tokens.append(("STRING", text[i + 1:j], i))
303
+ i = j + 1
304
+ continue
305
+ if ch.isalpha():
306
+ j = i
307
+ # `_` is a word CONTINUATION character, never a start: the
308
+ # name validators below decide legality (Has_bond_to is a
309
+ # predicate, c_o1 a constant), the tokenizer only spans the
310
+ # word. Stopping at `_` instead would split `Has_bond_to`
311
+ # into three tokens and mis-parse it.
312
+ while j < n and (text[j].isalnum() or text[j] == "_"):
313
+ j += 1
314
+ word = text[i:j]
315
+ if is_predicate_name(word):
316
+ tokens.append(("PRED", word, i))
317
+ elif is_referent(word):
318
+ tokens.append(("REF", word, i))
319
+ elif is_constant_name(word):
320
+ tokens.append(("CONST", word, i))
321
+ else:
322
+ raise DRSSyntaxError(
323
+ f"parse_drs: {word!r} at position {i} is not a legal referent, "
324
+ f"constant, or predicate name in {text!r}")
325
+ i = j
326
+ continue
327
+ raise DRSSyntaxError(f"parse_drs: unexpected character {ch!r} at position {i} in {text!r}")
328
+ tokens.append(("EOF", "", n))
329
+ return tokens
330
+
331
+
332
+ class _DRSParser:
333
+ """A single parse of one token stream; not re-used across calls."""
334
+
335
+ def __init__(self, tokens: List[_Token], text: str):
336
+ self._tokens = tokens
337
+ self._text = text
338
+ self._i = 0
339
+ self._mapping = NameMapping()
340
+
341
+ def _peek(self, ahead: int = 0) -> _Token:
342
+ return self._tokens[min(self._i + ahead, len(self._tokens) - 1)]
343
+
344
+ def _advance(self) -> _Token:
345
+ tok = self._tokens[self._i]
346
+ self._i += 1
347
+ return tok
348
+
349
+ def _error(self, message: str) -> DRSSyntaxError:
350
+ return DRSSyntaxError(f"parse_drs: {message} in {self._text!r}")
351
+
352
+ def _expect(self, ttype: str, what: str) -> _Token:
353
+ tok = self._peek()
354
+ if tok[0] != ttype:
355
+ found = "end of input" if tok[0] == "EOF" else f"{tok[1]!r}"
356
+ raise self._error(f"expected {what} but found {found} at position {tok[2]}")
357
+ return self._advance()
358
+
359
+ # -- grammar ---------------------------------------------------------- #
360
+
361
+ def parse_document(self) -> DRS:
362
+ result = self._boxexpr()
363
+ eof = self._peek()
364
+ if eof[0] != "EOF":
365
+ raise self._error(
366
+ f"unexpected trailing input {eof[1]!r} at position {eof[2]} (chaining "
367
+ f"multiple ->/∨/~ at the top level is not supported — nest boxes "
368
+ f"explicitly instead)")
369
+ if isinstance(result, DRS):
370
+ return result
371
+ return DRS((), (result,))
372
+
373
+ def _boxexpr(self) -> Union[DRS, Condition]:
374
+ if self._peek()[0] == "TILDE":
375
+ self._advance()
376
+ return Neg(self._box())
377
+ box = self._box()
378
+ nxt = self._peek()
379
+ if nxt[0] == "ARROW":
380
+ self._advance()
381
+ return Impl(box, self._box())
382
+ if nxt[0] == "OR":
383
+ self._advance()
384
+ return Or(box, self._box())
385
+ return box
386
+
387
+ def _box(self) -> DRS:
388
+ self._expect("LB", "'['")
389
+ refs = self._refs()
390
+ self._expect("PIPE", "'|' separating referents from conditions")
391
+ conds = self._conds()
392
+ self._expect("RB", "']'")
393
+ return DRS(tuple(refs), tuple(conds))
394
+
395
+ def _refs(self) -> List[str]:
396
+ if self._peek()[0] != "REF":
397
+ return []
398
+ out = [self._advance()[1]]
399
+ while self._peek()[0] == "COMMA":
400
+ self._advance()
401
+ out.append(self._expect("REF", "a referent name after ','")[1])
402
+ return out
403
+
404
+ def _conds(self) -> List[Condition]:
405
+ if self._peek()[0] == "RB":
406
+ return []
407
+ out = [self._cond()]
408
+ while self._peek()[0] == "COMMA":
409
+ self._advance()
410
+ out.append(self._cond())
411
+ return out
412
+
413
+ def _cond(self) -> Condition:
414
+ t = self._peek()
415
+ if t[0] == "TILDE":
416
+ self._advance()
417
+ return Neg(self._box())
418
+ if t[0] == "LB":
419
+ box = self._box()
420
+ nxt = self._peek()
421
+ if nxt[0] == "ARROW":
422
+ self._advance()
423
+ return Impl(box, self._box())
424
+ if nxt[0] == "OR":
425
+ self._advance()
426
+ return Or(box, self._box())
427
+ raise self._error(
428
+ f"a bare DRS ({box.to_box_notation()}) is not a valid condition by "
429
+ f"itself at position {t[2]} — wrap it in negation (~[...]) or make it "
430
+ f"one side of a duplex condition ([...] -> [...] / [...] ∨ [...])")
431
+ if t[0] == "PRED":
432
+ return self._pred_cond()
433
+ if t[0] in ("REF", "CONST", "STRING"):
434
+ return self._eq_cond()
435
+ raise self._error(
436
+ f"expected a condition (a predicate, an equality, ~[...], or "
437
+ f"[...] -> / ∨ [...]) but found "
438
+ f"{'end of input' if t[0] == 'EOF' else repr(t[1])} at position {t[2]}")
439
+
440
+ def _term(self) -> str:
441
+ t = self._peek()
442
+ if t[0] in ("REF", "CONST"):
443
+ self._advance()
444
+ return t[1]
445
+ if t[0] == "STRING":
446
+ self._advance()
447
+ return self._mapping.for_constant(t[1])
448
+ raise self._error(
449
+ f"expected a referent, constant, or quoted string but found "
450
+ f"{'end of input' if t[0] == 'EOF' else repr(t[1])} at position {t[2]}")
451
+
452
+ def _pred_cond(self) -> Condition:
453
+ name = self._advance()[1]
454
+ self._expect("LP", "'(' after the predicate name")
455
+ args = [self._term()]
456
+ if name == "Card" and self._peek()[0] == "COMMA":
457
+ # Card(g, >=, 3): the second element is a comparison OPERATOR,
458
+ # which no legal term can be -- so peeking one token past the
459
+ # comma decides Card-vs-Pred without ambiguity, and a Pred that
460
+ # happens to be NAMED Card (term-shaped args only) still parses.
461
+ if self._peek(1)[0] in ("CMP", "EQ"):
462
+ self._advance() # the comma
463
+ op = self._advance()[1]
464
+ self._expect("COMMA", "',' before the cardinality bound")
465
+ number = self._expect("NUMBER", "a non-negative integer bound")
466
+ self._expect("RP", "')'")
467
+ return Card(args[0], op, int(number[1]))
468
+ while self._peek()[0] == "COMMA":
469
+ self._advance()
470
+ args.append(self._term())
471
+ self._expect("RP", "')'")
472
+ if name == "Part_of":
473
+ # The typed membership condition (see nodes.Part): exactly two
474
+ # arguments, and a programmatically built Pred("Part_of", (m, g))
475
+ # re-parses as Part -- harmless on purpose, both export to the
476
+ # same Part_of atom.
477
+ if len(args) != 2:
478
+ raise self._error(
479
+ f"Part_of takes exactly two arguments (member, group), "
480
+ f"got {len(args)}")
481
+ return Part(args[0], args[1])
482
+ return Pred(name, tuple(args))
483
+
484
+ def _eq_cond(self) -> Eq:
485
+ a = self._term()
486
+ self._expect("EQ", "'=' (only a predicate application or an equality may "
487
+ "start with a referent/constant/string)")
488
+ b = self._term()
489
+ return Eq(a, b)
490
+
491
+
492
+ def parse_drs(text: str) -> DRS:
493
+ """Parse ``text`` (the compact box notation documented in the module docstring)
494
+ into a :class:`~unicode_logic_kit.drt.nodes.DRS`. Raises :class:`DRSSyntaxError` on
495
+ malformed input (unbalanced brackets, a bare box used as a condition, chained
496
+ ``->``/``∨``/``~``, an illegal referent/constant/predicate token, ...)."""
497
+ parser = _DRSParser(_tokenize_drs(text), text)
498
+ return parser.parse_document()
499
+
500
+
501
+ # =============================================================================
502
+ # 2. SBN subset.
503
+ # =============================================================================
504
+
505
+ _SENSE_RE = re.compile(r"^([a-z][a-z_]*)\.([nvars])\.([0-9]{2})$")
506
+ _ROLE_RE = re.compile(r"^[A-Z][a-zA-Z0-9]*(?:-[A-Z][a-zA-Z0-9]*)?$")
507
+ _OFFSET_RE = re.compile(r"^([+-])([0-9]+)$")
508
+ _ALLCAPS_RE = re.compile(r"^[A-Z_]+$")
509
+ _CONNECTOR_RE = re.compile(r"^([<>])([0-9]+)$")
510
+ _BARE_INTEGER_RE = re.compile(r"^[0-9]+$")
511
+
512
+ _NEGATION = "NEGATION"
513
+ #: The deictic-reference constants Bos (2023) §2.1 documents as bare (unquoted)
514
+ #: literals: the utterance time, its speaker, its addressee, and its location.
515
+ _DEICTIC_CONSTANTS = frozenset({"now", "speaker", "hearer", "here"})
516
+
517
+
518
+ @dataclass(frozen=True)
519
+ class SBNMapping:
520
+ """The mappings :func:`parse_sbn` used, so a lossy sanitization can be undone.
521
+
522
+ ``predicates`` maps each original sense token (``"person.n.01"``) to the predicate
523
+ name it became (``"PersonN01"``); ``constants`` maps each original literal constant
524
+ token — a (lower-cased) quoted string, or one of this subset's bare literals (a
525
+ deictic reference or a bare integer, accepted in either dialect — see the module
526
+ docstring's "Bare (unquoted) constants" bullet, part of the shared base grammar) —
527
+ to its sanitized constant name.
528
+ """
529
+
530
+ predicates: Dict[str, str]
531
+ constants: Dict[str, str]
532
+
533
+ def to_dict(self) -> dict:
534
+ return {"predicates": dict(self.predicates), "constants": dict(self.constants)}
535
+
536
+
537
+ def _sbn_predicate_name(token: str) -> str:
538
+ """Turn a WordNet-style sense token into a legal predicate name (see the module
539
+ docstring: ``person.n.01`` -> ``PersonN01``). Raises :class:`SBNSyntaxError` if
540
+ ``token`` is not a well-formed ``lemma.pos.NN`` sense token."""
541
+ m = _SENSE_RE.fullmatch(token)
542
+ if not m:
543
+ raise SBNSyntaxError(
544
+ f"parse_sbn: sense token {token!r} is not of the form 'lemma.pos.NN' "
545
+ f"(lowercase underscore-joined lemma, pos in n/v/a/r/s, two-digit sense) "
546
+ f"— this SBN subset only supports well-formed WordNet-style sense tokens.")
547
+ lemma, pos, sense = m.groups()
548
+ pascal = "".join(part[:1].upper() + part[1:] for part in lemma.split("_") if part)
549
+ return f"{pascal}{pos.upper()}{sense}"
550
+
551
+
552
+ def _sbn_role_predicate_name(role: str) -> str:
553
+ """A ROLE (or comparison/temporal/spatial OPERATOR token, e.g. ``EQU`` — lexically
554
+ indistinguishable from a role in this subset, see the module docstring) as a legal
555
+ kit predicate name. Each hyphen-segment of a VerbNet 'Co-' compound (``Co-Theme``,
556
+ :data:`_ROLE_RE`) is already uppercase-initial, so dropping the hyphen keeps it
557
+ PascalCase (``CoTheme``) — the kit PREDICATE convention has no hyphen. A no-op for
558
+ every role that has none, so this changes nothing for an already-accepted input."""
559
+ return role.replace("-", "")
560
+
561
+
562
+ def _split_respecting_quotes(s: str, lineno: int) -> List[str]:
563
+ """Whitespace-tokenize ``s``, keeping a quoted ``"..."`` span (which may itself
564
+ contain whitespace) as a single token including its quotes."""
565
+ tokens: List[str] = []
566
+ i, n = 0, len(s)
567
+ while i < n:
568
+ while i < n and s[i].isspace():
569
+ i += 1
570
+ if i >= n:
571
+ break
572
+ if s[i] == '"':
573
+ j = i + 1
574
+ while j < n and s[j] != '"':
575
+ j += 1
576
+ if j >= n:
577
+ raise SBNSyntaxError(f"parse_sbn: line {lineno}: unterminated quoted constant")
578
+ tokens.append(s[i:j + 1])
579
+ i = j + 1
580
+ else:
581
+ j = i
582
+ while j < n and not s[j].isspace():
583
+ j += 1
584
+ tokens.append(s[i:j])
585
+ i = j
586
+ return tokens
587
+
588
+
589
+ # Any line-break convention: a document handed over as a string may use CRLF
590
+ # or a bare CR, and a bare-CR one would otherwise be ONE line, its nesting lost.
591
+ _LINE_BREAK_RE = re.compile(r"\r\n?|\n")
592
+
593
+
594
+ def _split_sbn_lines(text: str) -> List[Tuple[str, int, str]]:
595
+ """Pass 1: strip comments (a PMB ``%%%``-prefixed generation-command header is just
596
+ another ``%...`` comment under this rule) and blank lines. Returns ``(content,
597
+ lineno, leading)`` triples, ``leading`` the RAW leading-whitespace substring —
598
+ whether it means TAB-only nesting depth or is pure cosmetic column alignment
599
+ depends on which of :func:`parse_sbn`'s two dialects the document uses (decided by
600
+ :func:`_uses_connector_dialect` once the whole document has been split), so it is
601
+ NOT validated here."""
602
+ out: List[Tuple[str, int, str]] = []
603
+ for lineno, raw in enumerate(_LINE_BREAK_RE.split(text), start=1):
604
+ line = raw.split("%", 1)[0]
605
+ if not line.strip():
606
+ continue
607
+ no_lead = line.lstrip(" \t")
608
+ leading = line[:len(line) - len(no_lead)]
609
+ out.append((no_lead.strip(), lineno, leading))
610
+ if not out:
611
+ raise SBNSyntaxError("parse_sbn: empty input (no content lines)")
612
+ return out
613
+
614
+
615
+ def _require_tab_indentation(lines: List[Tuple[str, int, str]]) -> None:
616
+ """The indentation dialect's own rule (see the module docstring): leading
617
+ whitespace is nesting depth and must be TAB-only. Raises on the first line that
618
+ isn't — unchanged from before the connector dialect existed."""
619
+ for _, lineno, leading in lines:
620
+ if " " in leading:
621
+ raise SBNSyntaxError(
622
+ f"parse_sbn: line {lineno} is indented with a space character; this "
623
+ f"SBN subset requires TAB-only indentation")
624
+
625
+
626
+ def _uses_connector_dialect(lines: List[Tuple[str, int, str]]) -> bool:
627
+ """True iff any box-operator line carries a trailing CONNECTOR token (``<N`` /
628
+ ``>N``) — decides which of :func:`parse_sbn`'s two dialects (see the module
629
+ docstring's "Dialect selection") to read the WHOLE document in. None of the
630
+ dialect-1 fixtures in ``tests/test_drt.py`` contain one, so they always resolve
631
+ False here, keeping their parse behaviour exactly as before."""
632
+ for content, _, _ in lines:
633
+ parts = content.split()
634
+ if (parts and _ALLCAPS_RE.fullmatch(parts[0])
635
+ and len(parts) == 2 and _CONNECTOR_RE.fullmatch(parts[1])):
636
+ return True
637
+ return False
638
+
639
+
640
+ @dataclass
641
+ class _SBNLine:
642
+ lineno: int
643
+ depth: int
644
+ index: int
645
+ kind: str # "sense" | "negation"
646
+ predicate: str = "" # sanitized predicate name, kind == "sense" only
647
+ role_targets: tuple = () # tuple of (role, raw_target) pairs
648
+
649
+
650
+ def parse_sbn(text: str) -> Tuple[DRS, SBNMapping]:
651
+ """Parse ``text`` into a ``(DRS, SBNMapping)`` pair, in whichever of the module
652
+ docstring's two documented SBN dialects ``text`` turns out to use (decided by
653
+ :func:`_uses_connector_dialect`). Raises :class:`SBNSyntaxError` on malformed
654
+ input or on any construct outside the chosen dialect's documented subset (naming
655
+ it explicitly)."""
656
+ raw_lines = _split_sbn_lines(text)
657
+ if _uses_connector_dialect(raw_lines):
658
+ return _parse_sbn_connector(raw_lines)
659
+ _require_tab_indentation(raw_lines)
660
+ return _parse_sbn_classic(raw_lines)
661
+
662
+
663
+ def _parse_sbn_classic(raw_lines: List[Tuple[str, int, str]]) -> Tuple[DRS, SBNMapping]:
664
+ """Dialect 1, the indentation dialect (see the module docstring): a bare
665
+ ``NEGATION`` line opens a sub-box via TAB depth. This is ``parse_sbn``'s ORIGINAL
666
+ subset — every input it already accepted parses to the identical DRS it always
667
+ did; the only additions since (widened ``_ROLE_RE``, the bare deictic/integer
668
+ constants below) are no-ops on any input that does not use them."""
669
+ # Pass 2: tokenize each line's own payload (sense/NEGATION keyword + role/target
670
+ # pairs), without resolving offset targets yet — that needs the full index->kind map.
671
+ pred_mapping: Dict[str, str] = {}
672
+ parsed_lines: List[_SBNLine] = []
673
+ for index, (content, lineno, leading) in enumerate(raw_lines, start=1):
674
+ depth = len(leading)
675
+ parts = _split_respecting_quotes(content, lineno)
676
+ head, rest = parts[0], parts[1:]
677
+ if head == _NEGATION:
678
+ if rest:
679
+ raise SBNSyntaxError(
680
+ f"parse_sbn: line {lineno}: 'NEGATION' takes no roles/targets in "
681
+ f"this subset, found {rest!r}")
682
+ parsed_lines.append(_SBNLine(lineno, depth, index, "negation"))
683
+ continue
684
+ if _ALLCAPS_RE.fullmatch(head):
685
+ raise SBNSyntaxError(
686
+ f"parse_sbn: line {lineno}: box-operator {head!r} is not supported by "
687
+ f"this SBN subset (only NEGATION is) — refusing rather than silently "
688
+ f"misreading it.")
689
+ if len(rest) % 2 != 0:
690
+ raise SBNSyntaxError(
691
+ f"parse_sbn: line {lineno}: role {rest[-1]!r} has no target")
692
+ role_targets = []
693
+ for k in range(0, len(rest), 2):
694
+ role, target = rest[k], rest[k + 1]
695
+ if not _ROLE_RE.fullmatch(role):
696
+ raise SBNSyntaxError(
697
+ f"parse_sbn: line {lineno}: role {role!r} must be uppercase-initial "
698
+ f"alphanumeric, optionally with one hyphenated segment (e.g. "
699
+ f"'Agent', 'Theme', 'Co-Theme')")
700
+ role_targets.append((role, target))
701
+ predicate = pred_mapping.get(head)
702
+ if predicate is None:
703
+ predicate = _sbn_predicate_name(head)
704
+ pred_mapping[head] = predicate
705
+ parsed_lines.append(_SBNLine(lineno, depth, index, "sense", predicate, tuple(role_targets)))
706
+
707
+ n = len(parsed_lines)
708
+ kind_by_index = {pl.index: pl.kind for pl in parsed_lines}
709
+
710
+ # Pass 3: resolve each role's raw target token to a final DRS term string.
711
+ const_mapping = NameMapping()
712
+ resolved: List[Tuple[_SBNLine, tuple]] = []
713
+ for pl in parsed_lines:
714
+ if pl.kind != "sense":
715
+ resolved.append((pl, ()))
716
+ continue
717
+ terms = []
718
+ for role, target in pl.role_targets:
719
+ m = _OFFSET_RE.fullmatch(target)
720
+ if m:
721
+ sign, digits = m.groups()
722
+ delta = int(digits) if sign == "+" else -int(digits)
723
+ target_index = pl.index + delta
724
+ if target_index < 1 or target_index > n:
725
+ raise SBNSyntaxError(
726
+ f"parse_sbn: line {pl.lineno}: offset {target!r} on role "
727
+ f"{role!r} points to line index {target_index}, outside the "
728
+ f"document (1..{n})")
729
+ if kind_by_index[target_index] != "sense":
730
+ raise SBNSyntaxError(
731
+ f"parse_sbn: line {pl.lineno}: offset {target!r} on role "
732
+ f"{role!r} targets line {target_index}, a box-operator line "
733
+ f"with no referent of its own in this subset")
734
+ terms.append((role, f"e{target_index}"))
735
+ elif target.startswith('"') and target.endswith('"') and len(target) >= 2:
736
+ raw_const = target[1:-1].lower()
737
+ terms.append((role, const_mapping.for_constant(raw_const)))
738
+ elif target in _DEICTIC_CONSTANTS or _BARE_INTEGER_RE.fullmatch(target):
739
+ terms.append((role, const_mapping.for_constant(target)))
740
+ else:
741
+ raise SBNSyntaxError(
742
+ f"parse_sbn: line {pl.lineno}: target {target!r} on role {role!r} "
743
+ f"is neither a signed line offset (+N / -N), a quoted constant "
744
+ f'("..."), nor one of this subset\'s bare constant literals '
745
+ f"(now/speaker/hearer/here, or a bare integer).")
746
+ resolved.append((pl, tuple(terms)))
747
+
748
+ # Pass 4: build the nested DRS via an indentation-driven stack of open boxes.
749
+ class _Frame:
750
+ __slots__ = ("depth", "referents", "conditions")
751
+
752
+ def __init__(self, depth: int):
753
+ self.depth = depth
754
+ self.referents: List[str] = []
755
+ self.conditions: List[Condition] = []
756
+
757
+ def _close(frame: "_Frame") -> DRS:
758
+ return DRS(tuple(frame.referents), tuple(frame.conditions))
759
+
760
+ def _pop_negation(stack: List["_Frame"]) -> None:
761
+ closed = stack.pop()
762
+ if not closed.referents and not closed.conditions:
763
+ raise SBNSyntaxError(
764
+ "parse_sbn: a NEGATION line has no content — every NEGATION in this "
765
+ "subset must be followed by at least one more-indented line")
766
+ stack[-1].conditions.append(Neg(_close(closed)))
767
+
768
+ stack: List[_Frame] = [_Frame(0)]
769
+ for pl, terms in resolved:
770
+ d = pl.depth
771
+ while len(stack) > 1 and stack[-1].depth > d:
772
+ _pop_negation(stack)
773
+ current = stack[-1]
774
+ if current.depth != d:
775
+ raise SBNSyntaxError(
776
+ f"parse_sbn: line {pl.lineno}: indentation depth {d} does not open or "
777
+ f"continue any box (expected depth {current.depth})")
778
+ if pl.kind == "negation":
779
+ stack.append(_Frame(d + 1))
780
+ else:
781
+ ref = f"e{pl.index}"
782
+ current.referents.append(ref)
783
+ current.conditions.append(Pred(pl.predicate, (ref,)))
784
+ for role, term in terms:
785
+ current.conditions.append(Pred(_sbn_role_predicate_name(role), (ref, term)))
786
+
787
+ while len(stack) > 1:
788
+ _pop_negation(stack)
789
+
790
+ root = _close(stack[0])
791
+ # Offsets are only range/kind-checked above; an offset that crosses a
792
+ # NEGATION box boundary builds a structurally well-formed but
793
+ # ACCESSIBILITY-violating DRS (review-flagged: a referent used outside
794
+ # the box that declares it). Validate before handing it out so a caller
795
+ # never receives a silently invalid tree — the violation surfaces here,
796
+ # named, instead of downstream in drs_to_fol. nodes.DRS.validate raises a
797
+ # plain ValueError (it has no SBN-specific vocabulary of its own); re-raise
798
+ # as SBNSyntaxError so every parse_sbn refusal is uniformly that one type.
799
+ try:
800
+ root.validate()
801
+ except ValueError as e:
802
+ raise SBNSyntaxError(str(e)) from e
803
+ mapping = SBNMapping(predicates=dict(pred_mapping), constants=dict(const_mapping.constant))
804
+ return root, mapping
805
+
806
+
807
+ def _parse_sbn_connector(raw_lines: List[Tuple[str, int, str]]) -> Tuple[DRS, SBNMapping]:
808
+ """Dialect 2, the connector dialect (see the module docstring): PMB's own
809
+ released SBN format. Leading whitespace is ignored throughout; box nesting comes
810
+ from each box-operator line's trailing connector (``<N``) rather than
811
+ indentation, and role-hook indices count CONCEPT lines only."""
812
+
813
+ class _Context:
814
+ __slots__ = ("referents", "slots")
815
+
816
+ def __init__(self) -> None:
817
+ self.referents: List[str] = []
818
+ # Each slot is either a ("concept", i) placeholder (i indexes `concepts`
819
+ # below) or a ("neg", child_context_index) placeholder — both resolved
820
+ # into actual Conditions only once every concept's offsets and every
821
+ # child context are known (see the two passes below).
822
+ self.slots: List[Tuple[str, int]] = []
823
+
824
+ contexts: List[_Context] = [_Context()] # context 0 = the outermost, implicit context
825
+ neg_lineno: Dict[int, int] = {} # child context index -> its NEGATION's line
826
+ current = 0
827
+ pred_mapping: Dict[str, str] = {}
828
+ const_mapping = NameMapping()
829
+ concepts: List[dict] = [] # concept-only registry, in document order
830
+
831
+ # Pass 1: one linear scan building the context graph and every concept's raw
832
+ # (unresolved) role targets — an offset may point FORWARD to a concept not yet
833
+ # seen, so resolution is deferred to pass 2 below.
834
+ for content, lineno, _leading in raw_lines:
835
+ head = content.split(None, 1)[0]
836
+ if _ALLCAPS_RE.fullmatch(head):
837
+ rest = content.split()[1:]
838
+ if len(rest) != 1 or not _CONNECTOR_RE.fullmatch(rest[0]):
839
+ raise SBNSyntaxError(
840
+ f"parse_sbn: line {lineno}: a box-operator line in the connector "
841
+ f"dialect (see the module docstring) must be followed by exactly "
842
+ f"one connector (e.g. 'NEGATION <1'), found {rest!r}")
843
+ sign, digits = rest[0][0], rest[0][1:]
844
+ if sign == ">":
845
+ raise SBNSyntaxError(
846
+ f"parse_sbn: line {lineno}: a forward connector ({rest[0]!r}) is "
847
+ f"not supported by this SBN subset — only backward ('<N') "
848
+ f"connectors are.")
849
+ if head != _NEGATION:
850
+ raise SBNSyntaxError(
851
+ f"parse_sbn: line {lineno}: box-operator {head!r} is not "
852
+ f"supported by this SBN subset (only NEGATION is) — refusing "
853
+ f"rather than silently misreading it.")
854
+ n = int(digits)
855
+ new_index = len(contexts)
856
+ target_index = new_index - n
857
+ if n < 1 or target_index < 0:
858
+ raise SBNSyntaxError(
859
+ f"parse_sbn: line {lineno}: connector '<{n}' refers to context "
860
+ f"{target_index}, outside the document (contexts "
861
+ f"0..{new_index - 1} exist at this point)")
862
+ contexts.append(_Context())
863
+ contexts[target_index].slots.append(("neg", new_index))
864
+ neg_lineno[new_index] = lineno
865
+ current = new_index
866
+ continue
867
+
868
+ # A concept (sense) line — tokenized quote-aware (unlike the box-operator
869
+ # check above, a concept's role targets may be quoted strings).
870
+ toks = _split_respecting_quotes(content, lineno)
871
+ head, rest = toks[0], toks[1:]
872
+ if len(rest) % 2 != 0:
873
+ raise SBNSyntaxError(
874
+ f"parse_sbn: line {lineno}: role {rest[-1]!r} has no target")
875
+ role_targets = []
876
+ for k in range(0, len(rest), 2):
877
+ role, target = rest[k], rest[k + 1]
878
+ if not _ROLE_RE.fullmatch(role):
879
+ raise SBNSyntaxError(
880
+ f"parse_sbn: line {lineno}: role {role!r} must be uppercase-initial "
881
+ f"alphanumeric, optionally with one hyphenated segment (e.g. "
882
+ f"'Agent', 'Theme', 'Co-Theme')")
883
+ role_targets.append((role, target))
884
+ predicate = pred_mapping.get(head)
885
+ if predicate is None:
886
+ predicate = _sbn_predicate_name(head)
887
+ pred_mapping[head] = predicate
888
+ concept_index = len(concepts) # 0-based position
889
+ ref = f"e{concept_index + 1}"
890
+ contexts[current].referents.append(ref)
891
+ contexts[current].slots.append(("concept", concept_index))
892
+ concepts.append({"lineno": lineno, "ref": ref, "predicate": predicate,
893
+ "role_targets": role_targets, "resolved": None})
894
+
895
+ n_concepts = len(concepts)
896
+
897
+ # Pass 2: resolve every role-hook target now that each concept's CONCEPT-ONLY
898
+ # position (1-based, box-operator lines excluded — see the module docstring) is
899
+ # known.
900
+ for i, c in enumerate(concepts, start=1):
901
+ resolved_terms = []
902
+ for role, target in c["role_targets"]:
903
+ m = _OFFSET_RE.fullmatch(target)
904
+ if m:
905
+ sign, digits = m.groups()
906
+ delta = int(digits) if sign == "+" else -int(digits)
907
+ target_index = i + delta
908
+ if target_index < 1 or target_index > n_concepts:
909
+ raise SBNSyntaxError(
910
+ f"parse_sbn: line {c['lineno']}: offset {target!r} on role "
911
+ f"{role!r} points to concept index {target_index}, outside "
912
+ f"the document ({n_concepts} concept(s))")
913
+ resolved_terms.append((role, concepts[target_index - 1]["ref"]))
914
+ elif target.startswith('"') and target.endswith('"') and len(target) >= 2:
915
+ raw_const = target[1:-1].lower()
916
+ resolved_terms.append((role, const_mapping.for_constant(raw_const)))
917
+ elif target in _DEICTIC_CONSTANTS or _BARE_INTEGER_RE.fullmatch(target):
918
+ resolved_terms.append((role, const_mapping.for_constant(target)))
919
+ elif _CONNECTOR_RE.fullmatch(target):
920
+ raise SBNSyntaxError(
921
+ f"parse_sbn: line {c['lineno']}: role {role!r}'s target "
922
+ f"{target!r} is a context/box reference (an embedded-clause or "
923
+ f"propositional-attitude argument) — this SBN subset does not "
924
+ f"support box-valued role targets.")
925
+ else:
926
+ raise SBNSyntaxError(
927
+ f"parse_sbn: line {c['lineno']}: target {target!r} on role "
928
+ f"{role!r} is neither a signed concept offset (+N / -N), a "
929
+ f'quoted constant ("..."), nor one of this subset\'s bare '
930
+ f"constant literals (now/speaker/hearer/here, or a bare integer).")
931
+ c["resolved"] = resolved_terms
932
+
933
+ # Pass 3: close every context bottom-up. A "neg" slot always names a context
934
+ # with a STRICTLY GREATER index than its own (target_index < new_index above),
935
+ # so resolving from the highest index down guarantees a child is already closed
936
+ # by the time its parent needs it.
937
+ closed: Dict[int, DRS] = {}
938
+ for k in range(len(contexts) - 1, -1, -1):
939
+ conditions: List[Condition] = []
940
+ for kind, idx in contexts[k].slots:
941
+ if kind == "concept":
942
+ c = concepts[idx]
943
+ conditions.append(Pred(c["predicate"], (c["ref"],)))
944
+ for role, term in c["resolved"]:
945
+ conditions.append(Pred(_sbn_role_predicate_name(role), (c["ref"], term)))
946
+ else: # "neg"
947
+ conditions.append(Neg(closed[idx]))
948
+ if k != 0 and not contexts[k].referents and not conditions:
949
+ raise SBNSyntaxError(
950
+ f"parse_sbn: line {neg_lineno[k]}: NEGATION introduces an empty "
951
+ f"context — every NEGATION in this subset must be followed by at "
952
+ f"least one concept before the next box-operator or the end of the "
953
+ f"document")
954
+ closed[k] = DRS(tuple(contexts[k].referents), tuple(conditions))
955
+
956
+ root = closed[0]
957
+ # See the identical comment in _parse_sbn_classic above: validated here rather
958
+ # than left for drs_to_fol to discover downstream, and re-raised as
959
+ # SBNSyntaxError so every parse_sbn refusal is uniformly that one type.
960
+ try:
961
+ root.validate()
962
+ except ValueError as e:
963
+ raise SBNSyntaxError(str(e)) from e
964
+ mapping = SBNMapping(predicates=dict(pred_mapping), constants=dict(const_mapping.constant))
965
+ return root, mapping