unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,965 @@
|
|
|
1
|
+
"""Two parsers into :class:`~unicode_logic_kit.drt.nodes.DRS`: a compact hand-rolled box
|
|
2
|
+
notation (:func:`parse_drs`), and a documented SUBSET of the Parallel Meaning Bank's
|
|
3
|
+
Sequence Box Notation (:func:`parse_sbn`) — the PMB's line-based interchange format for
|
|
4
|
+
DRS-like structures (van Noord et al.'s Parallel Meaning Bank project releases sense-
|
|
5
|
+
disambiguated, role-annotated semantic parses in this format; it is the natural "found
|
|
6
|
+
data" source for DRSs, hence the SBN import path).
|
|
7
|
+
|
|
8
|
+
Both are small, purpose-built, single-shot grammars — no dialect sharing, no term/lambda
|
|
9
|
+
layer — so, exactly as ``unicode_logic_kit.dl.parser`` argues for ALC concepts, a hand-rolled
|
|
10
|
+
tokenizer/recursive-descent parser is simpler and gives better per-construct error messages
|
|
11
|
+
than standing up a Lark grammar + transformer for either of them.
|
|
12
|
+
|
|
13
|
+
=================================================================================
|
|
14
|
+
1. The box notation (``parse_drs``)
|
|
15
|
+
=================================================================================
|
|
16
|
+
|
|
17
|
+
Grammar (``|`` = alternative, ``?`` = optional, ``*`` = zero-or-more; NAME classes are
|
|
18
|
+
exactly ``unicode_logic_kit.drt.nodes``'s ``is_referent`` / ``is_predicate_name`` /
|
|
19
|
+
``is_constant_name``, i.e. the kit's own VARIABLE / PREDICATE / NAME lexical conventions)::
|
|
20
|
+
|
|
21
|
+
document := boxexpr EOF
|
|
22
|
+
boxexpr := "~" box # -> Neg(box)
|
|
23
|
+
| box (("->" | "∨") box)? # -> box | Impl(box, box) | Or(box, box)
|
|
24
|
+
|
|
25
|
+
box := "[" refs? "|" conds? "]"
|
|
26
|
+
refs := REF ("," REF)*
|
|
27
|
+
conds := cond ("," cond)*
|
|
28
|
+
cond := pred_cond | eq_cond
|
|
29
|
+
| "~" box # -> Neg(box)
|
|
30
|
+
| box ("->" | "∨") box # -> Impl(box, box) | Or(box, box)
|
|
31
|
+
|
|
32
|
+
pred_cond := PRED "(" term ("," term)* ")"
|
|
33
|
+
| "Card" "(" term "," CMP "," NUMBER ")" # -> Card (0.24.0; the operator
|
|
34
|
+
| # slot disambiguates against
|
|
35
|
+
| # a Pred merely NAMED Card)
|
|
36
|
+
| "Part_of" "(" term "," term ")" # -> Part (0.24.0; exactly 2 args)
|
|
37
|
+
eq_cond := term "=" term
|
|
38
|
+
term := REF | CONST | STRING
|
|
39
|
+
CMP := ">=" | "<=" | "=" | ">" | "<"
|
|
40
|
+
NUMBER := /[0-9]+/
|
|
41
|
+
|
|
42
|
+
REF : a single lowercase letter optionally followed by digits (x, y, e12, ...)
|
|
43
|
+
CONST : a lowercase-initial bare identifier of >= 2 characters (john, daisy, ...)
|
|
44
|
+
PRED : an uppercase-initial identifier, underscores allowed in
|
|
45
|
+
continuation (Farmer, Owns, Has_bond_to, ...)
|
|
46
|
+
STRING : a double-quoted literal ("John Doe") — sanitized to a legal CONST via
|
|
47
|
+
unicode_logic_kit.fol.sanitize.NameMapping (kept consistent within one
|
|
48
|
+
parse_drs call; not returned — see "Quoted constants" below)
|
|
49
|
+
|
|
50
|
+
Worked example (the classic donkey sentence, exactly as it appears in the kit roadmap)::
|
|
51
|
+
|
|
52
|
+
[x, y | Farmer(x), Donkey(y), Owns(x, y)] -> [ | Beats(x, y)]
|
|
53
|
+
|
|
54
|
+
parses to ``DRS((), (Impl(DRS(("x","y"), (Pred("Farmer",("x",)), Pred("Donkey",("y",)),
|
|
55
|
+
Pred("Owns",("x","y")))), DRS((), (Pred("Beats",("x","y")),))),))`` — a top-level bare
|
|
56
|
+
``Impl``/``Neg``/``Or`` result is wrapped as the sole condition of an empty-referent DRS
|
|
57
|
+
(a "bare box" result, with no trailing ``->``/``∨``, IS the returned DRS directly).
|
|
58
|
+
|
|
59
|
+
**Deliberately unambiguous, at the cost of expressiveness.** ``boxexpr`` accepts AT MOST
|
|
60
|
+
ONE ``->``/``∨`` after a box, and ``~`` only ever prefixes a bare ``box`` (never a whole
|
|
61
|
+
``boxexpr``) — so ``~[...] -> [...]`` and ``[...] -> [...] -> [...]`` are BOTH refused
|
|
62
|
+
(with a message naming the dangling trailing token) rather than silently picking a
|
|
63
|
+
left/right-associativity or an operator-precedence convention nobody asked for. Nest boxes
|
|
64
|
+
explicitly instead — e.g. ``[ | ~[...]] -> [...]`` for the first case — box notation has
|
|
65
|
+
no expressiveness loss, only a syntax-level one. A "cond" that is a bare box with no
|
|
66
|
+
``->``/``∨`` suffix is refused too (naming the position): a bare DRS is not one of the five
|
|
67
|
+
condition variants (:mod:`unicode_logic_kit.drt.nodes`) — wrap it in ``~[...]`` or make it a
|
|
68
|
+
``->``/``∨`` operand.
|
|
69
|
+
|
|
70
|
+
**Quoted constants.** A bare CONST token is already legal (``is_constant_name``) and passes
|
|
71
|
+
through unchanged. A quoted ``"..."`` STRING is for constants box notation's bare-token
|
|
72
|
+
syntax cannot express as-is (spaces, punctuation, digit-leading, single characters that
|
|
73
|
+
would otherwise collide with the referent namespace, non-ASCII, ...); it is sanitized via a
|
|
74
|
+
fresh :class:`~unicode_logic_kit.fol.sanitize.NameMapping` PER ``parse_drs`` CALL (consistent
|
|
75
|
+
within one call — the same quoted string always sanitizes to the same constant — but not
|
|
76
|
+
returned to the caller, since the primary, recommended path is to just write a legal bare
|
|
77
|
+
CONST directly, as the roadmap's own donkey-sentence facts do: ``Farmer(john)``, no quotes
|
|
78
|
+
needed). :func:`parse_sbn` below, whose quoted constants are the norm rather than an
|
|
79
|
+
escape hatch, DOES return its mapping.
|
|
80
|
+
|
|
81
|
+
=================================================================================
|
|
82
|
+
2. The SBN subset (``parse_sbn``)
|
|
83
|
+
=================================================================================
|
|
84
|
+
|
|
85
|
+
Real SBN is a line-per-token format: each line names a WordNet-style sense for one token of
|
|
86
|
+
the sentence, optionally followed by semantic-role edges to OTHER lines (by relative line
|
|
87
|
+
offset) or to literal constants. This function implements a precisely bounded SUBSET,
|
|
88
|
+
documented here in full — anything outside it is REFUSED, naming the construct, rather than
|
|
89
|
+
silently mis-parsed::
|
|
90
|
+
|
|
91
|
+
sbn := line+
|
|
92
|
+
line := TABS (sense_line | "NEGATION") COMMENT?
|
|
93
|
+
sense_line := SENSE (ROLE target)*
|
|
94
|
+
SENSE := LEMMA "." POS "." SENSE_NUM # e.g. farmer.n.01
|
|
95
|
+
LEMMA : /[a-z][a-z_]*/ # underscore-joined lowercase words
|
|
96
|
+
POS : one of n, v, a, r, s
|
|
97
|
+
SENSE_NUM : exactly two digits
|
|
98
|
+
ROLE := /[A-Z][a-zA-Z0-9]*(-[A-Z][a-zA-Z0-9]*)?/ # e.g. Agent, Patient, Co-Theme
|
|
99
|
+
target := OFFSET | STRING
|
|
100
|
+
OFFSET : /[+-][0-9]+/ # relative CONTENT-line index (see below)
|
|
101
|
+
STRING : a double-quoted literal
|
|
102
|
+
COMMENT := "%" rest-of-line, ignored wherever it appears
|
|
103
|
+
blank (whitespace-only) lines are ignored and do not consume a content-line index.
|
|
104
|
+
|
|
105
|
+
* **Indentation is exactly TAB characters** (one tab = one nesting depth; mixing in spaces,
|
|
106
|
+
or indenting with spaces at all, is refused). Depth 0 lines form the outermost DRS; a
|
|
107
|
+
depth-``d`` line may only be followed by a depth-``d+1`` line if it is a ``NEGATION``
|
|
108
|
+
line (opening a sub-box that the following, deeper-indented lines belong to, up to but
|
|
109
|
+
not including the next line at depth <= d, or EOF) — any other depth jump (skipping a
|
|
110
|
+
level, or dedenting to a depth that was never open) is refused, naming the line.
|
|
111
|
+
* **Line numbering.** Every CONTENT line (a ``sense_line`` or a ``NEGATION`` line — blank
|
|
112
|
+
and comment-only lines do not count) gets a 1-based index in document order, REGARDLESS
|
|
113
|
+
of indentation depth; a ``sense_line`` at index ``i`` introduces the referent ``f"e{i}"``.
|
|
114
|
+
A ``NEGATION`` line consumes an index too (so offsets past it stay stable) but introduces
|
|
115
|
+
no referent of its own — targeting one is refused, naming the line ("a box-operator line
|
|
116
|
+
has no referent in this subset").
|
|
117
|
+
* **Sense -> predicate, exactly as the kit roadmap specifies**: ``person.n.01`` becomes the
|
|
118
|
+
predicate ``PersonN01`` (each dot-separated part capitalised — an underscore-joined lemma
|
|
119
|
+
like ``get_up.v.02`` becomes ``GetUpV02`` — and concatenated; PoS and sense-number keep
|
|
120
|
+
their literal digits/letter). The token -> predicate mapping is recorded and returned
|
|
121
|
+
(:class:`SBNMapping.predicates`) since it is lossy (case and the dots are gone).
|
|
122
|
+
* **Role -> binary predicate**, exactly as written (``Agent``) applied to
|
|
123
|
+
``(this_line's_referent, target)`` — i.e. ``Agent(e3, e1)`` style, per the roadmap.
|
|
124
|
+
**A role name may carry one hyphenated segment** (``Co-Theme``, ``Co-Agent``,
|
|
125
|
+
``Co-Patient``, ...) — VerbNet's own "Co-" compounding, common throughout real PMB
|
|
126
|
+
releases; the hyphen is dropped when the role becomes a predicate name (``Co-Theme``
|
|
127
|
+
-> ``CoTheme``, since the kit PREDICATE convention has no hyphen). This is part of the
|
|
128
|
+
BASE grammar above, in force in both SBN dialects this function reads (see 2b below) —
|
|
129
|
+
real PMB documents use ``Co-``-compounded roles whether or not they also happen to use
|
|
130
|
+
the connector dialect's own box-operator mechanism (e.g. a flat document with no
|
|
131
|
+
``NEGATION`` line at all can still carry a ``Co-Theme`` role), so this widening cannot
|
|
132
|
+
be gated to one dialect without silently refusing real, in-scope input.
|
|
133
|
+
* **The only supported box-changing operator is ``NEGATION``.** Any other ALL-CAPS token
|
|
134
|
+
in sense-token position (``POSSIBLE``, ``NECESSARY``, ``DISCOURSE_REFERENCE``, PMB's
|
|
135
|
+
other real operators, ...) is REFUSED BY NAME: this subset is negation-only. A
|
|
136
|
+
``NEGATION`` line with no more-indented line following it (an empty scope) is refused too
|
|
137
|
+
— in this subset a box operator only ever appears to introduce content, never vacuously.
|
|
138
|
+
* **Quoted constants** are sanitized via :class:`~unicode_logic_kit.fol.sanitize.NameMapping`
|
|
139
|
+
(lower-cased first, so ``"John"`` and ``"john"`` map together), and the mapping IS
|
|
140
|
+
returned (:class:`SBNMapping.constants`) — unlike the box notation's escape-hatch
|
|
141
|
+
quoting, a quoted literal is SBN's ONLY way to write a constant, so recovering the
|
|
142
|
+
original matters.
|
|
143
|
+
* **Bare (unquoted) constants**: the four deictic references Bos (2023) §2.1 documents
|
|
144
|
+
(``now``/``speaker``/``hearer``/``here`` — utterance time/speaker/addressee/location) and
|
|
145
|
+
a bare UNSIGNED integer (``Quantity 3``) are also accepted as constant targets, routed
|
|
146
|
+
through the same :class:`NameMapping` as quoted constants. Unambiguous with an OFFSET
|
|
147
|
+
target, which always carries a sign.
|
|
148
|
+
* **A comment may itself contain a ``%``, ANSI colour escapes, or start with the PMB release
|
|
149
|
+
format's own ``%%%``-prefixed generation-command header** — all of that is already inside
|
|
150
|
+
the comment by the ``COMMENT`` rule above (everything from the first ``%`` on the line),
|
|
151
|
+
so none of it needs special handling.
|
|
152
|
+
|
|
153
|
+
**What is explicitly out of scope** (refused by name, never silently dropped): event
|
|
154
|
+
quantification / plural referents, presupposition triggers, any operator besides NEGATION,
|
|
155
|
+
multi-word discourse (this parses ONE sbn "document" — i.e. one sentence's worth of boxes —
|
|
156
|
+
per call, matching this kit subpackage's single-sentence-plus-anaphora scope; see the
|
|
157
|
+
``unicode_logic_kit.drt`` package docstring).
|
|
158
|
+
|
|
159
|
+
=================================================================================
|
|
160
|
+
2b. A second SBN dialect: PMB's own released format (the "connector" dialect)
|
|
161
|
+
=================================================================================
|
|
162
|
+
|
|
163
|
+
The dialect above was hand-written before any real PMB release was measured against it.
|
|
164
|
+
PMB's actual gold releases (verified against pmb-5.1.0's
|
|
165
|
+
``data/<lang>/gold/p<NN>/d<NNNN>/<lang>.drs.sbn`` files) use a DIFFERENT, but fully
|
|
166
|
+
documented, mechanism instead of textual indentation — Bos (2023), "The Sequence Notation:
|
|
167
|
+
Catching Complex Meanings in Simple Graphs" (IWCS 2023), §§2.1/3.4/4.1.
|
|
168
|
+
:func:`parse_sbn` recognizes it automatically (see "Dialect selection" below) and reads it
|
|
169
|
+
as follows — everything not listed here (LEMMA/POS/SENSE_NUM, roles, quoted constants,
|
|
170
|
+
comments, ...) is exactly as in the dialect above:
|
|
171
|
+
|
|
172
|
+
* **Leading whitespace is PURE COSMETIC COLUMN ALIGNMENT**, never nesting depth (verified:
|
|
173
|
+
no CONCEPT line in pmb-5.1.0 ever carries leading whitespace; only box-operator lines do,
|
|
174
|
+
padding them to the sense-token column for human readability) — this dialect ignores it
|
|
175
|
+
entirely, on every line, and never refuses a space the indentation dialect above would.
|
|
176
|
+
* **A box-operator line carries a trailing CONNECTOR token** (``<N``, a positive integer)
|
|
177
|
+
instead of relying on indentation. Contexts (boxes) are numbered by INTRODUCTION ORDER:
|
|
178
|
+
context 0 is the outermost, implicit context holding everything before the first
|
|
179
|
+
separator; the K-th separator encountered (1-based, document order) introduces context K.
|
|
180
|
+
Its connector ``<N`` (``1 <= N <= K``) says the separator's OWN reading (``Neg`` for
|
|
181
|
+
``NEGATION``) is a CONDITION of context ``K - N`` — e.g. ``<1`` attaches to the immediately
|
|
182
|
+
preceding context, ``<2`` skips one further back, and so on; several separators may attach
|
|
183
|
+
to the SAME earlier context (e.g. "she is neither rich nor famous": ``NEGATION <1`` then
|
|
184
|
+
``NEGATION <2``, both landing on context 0, giving ``¬Rich(x) ∧ ¬Famous(x)`` rather than a
|
|
185
|
+
nested double negation — Bos 2023 Figure 4). As above, only ``NEGATION`` is a supported
|
|
186
|
+
separator; every other name PMB emits (``POSSIBILITY``, ``NECESSITY``, and the SDRT
|
|
187
|
+
discourse relations ``CONTINUATION``, ``CONTRAST``, ``CONJUNCTION``, ...) is refused by
|
|
188
|
+
name — this subset's DRS conditions have no modal-box or discourse-relation reading for
|
|
189
|
+
them. A FORWARD connector (``>N``) is refused too (it needs two-pass resolution this
|
|
190
|
+
subset does not implement).
|
|
191
|
+
* **Role-hook indices (``+N``/``-N``) count CONCEPT (sense) lines ONLY**, skipping every
|
|
192
|
+
box-operator line entirely — DELIBERATELY DIFFERENT from the dialect above's own
|
|
193
|
+
numbering (which counts box-operator lines too, so that offsets stay stable across them):
|
|
194
|
+
this is what Bos (2023) §3.3 and pmb-5.1.0 itself actually implement. Changing the other
|
|
195
|
+
dialect's existing (self-consistent, if non-standard) numbering would break its own
|
|
196
|
+
already-accepted inputs, so both numbering conventions coexist, one per dialect.
|
|
197
|
+
* A role target that is itself a connector (``Proposition >1`` — an embedded-clause /
|
|
198
|
+
propositional-attitude argument pointing AT A CONTEXT rather than a concept) is refused by
|
|
199
|
+
name: this subset only resolves entity-valued role targets.
|
|
200
|
+
* Role names may carry one hyphenated segment (``Co-Theme``, ``Co-Agent``, ...) — this is
|
|
201
|
+
the BASE grammar's own rule (see the ``ROLE`` widening in section 2 above), not a
|
|
202
|
+
connector-dialect addition; it is repeated here only because ``Co-`` compounding is
|
|
203
|
+
especially common throughout pmb-5.1.0's connector-dialect documents. Comparison/
|
|
204
|
+
temporal/spatial OPERATOR tokens (``EQU``, ``TPR``, ...) are lexically indistinguishable
|
|
205
|
+
from roles in this subset and get the SAME uninterpreted binary-predicate treatment —
|
|
206
|
+
none of their axiomatic content (e.g. ``EQU``'s reflexivity) is asserted, only the atom
|
|
207
|
+
itself.
|
|
208
|
+
|
|
209
|
+
**Dialect selection.** :func:`parse_sbn` looks at every box-operator line up front: if ANY
|
|
210
|
+
of them carries a trailing connector token, the WHOLE document is read in the connector
|
|
211
|
+
dialect (every box-operator line must then carry one, or the specific line is named and
|
|
212
|
+
refused); otherwise it is read in the indentation dialect above (a bare ``NEGATION``, TAB
|
|
213
|
+
depth). The two dialects' numbering/indentation conventions are per-document, never mixed
|
|
214
|
+
within one call — this is exactly why every existing ``parse_sbn`` input (none of which
|
|
215
|
+
contains a connector token) keeps parsing to the identical DRS it always has.
|
|
216
|
+
|
|
217
|
+
The 2-3 SBN examples exercised in ``tests/test_drt.py`` for the indentation dialect are
|
|
218
|
+
CONSTRUCTED BY HAND — they are illustrative, not verbatim PMB corpus data. The connector
|
|
219
|
+
dialect is exercised against both hand-written fixtures and, when a caller points
|
|
220
|
+
``UFK_PMB_SBN_DIR`` at one, a real PMB gold release — see
|
|
221
|
+
:mod:`unicode_logic_kit.eval.datasets.pmb` and ``tests/test_datasets_pmb.py``.
|
|
222
|
+
"""
|
|
223
|
+
|
|
224
|
+
import re
|
|
225
|
+
from dataclasses import dataclass
|
|
226
|
+
from typing import Dict, List, Tuple, Union
|
|
227
|
+
|
|
228
|
+
from ..fol.sanitize import NameMapping
|
|
229
|
+
from .nodes import (
|
|
230
|
+
DRS, Card, Condition, Pred, Eq, Neg, Impl, Or, Part,
|
|
231
|
+
is_referent, is_predicate_name, is_constant_name,
|
|
232
|
+
)
|
|
233
|
+
|
|
234
|
+
__all__ = [
|
|
235
|
+
"parse_drs", "DRSSyntaxError",
|
|
236
|
+
"parse_sbn", "SBNSyntaxError", "SBNMapping",
|
|
237
|
+
]
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
class DRSSyntaxError(ValueError):
|
|
241
|
+
"""Raised by :func:`parse_drs` on malformed box-notation input."""
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
class SBNSyntaxError(ValueError):
|
|
245
|
+
"""Raised by :func:`parse_sbn` on malformed input, or a construct outside the
|
|
246
|
+
documented SBN subset (see the module docstring)."""
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
# =============================================================================
|
|
250
|
+
# 1. Box notation.
|
|
251
|
+
# =============================================================================
|
|
252
|
+
|
|
253
|
+
_GLYPHS = {
|
|
254
|
+
"[": "LB", "]": "RB", "|": "PIPE", ",": "COMMA", "~": "TILDE",
|
|
255
|
+
"=": "EQ", "(": "LP", ")": "RP", "∨": "OR",
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
# A Token is (type: str, value: str, pos: int).
|
|
259
|
+
_Token = Tuple[str, str, int]
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def _tokenize_drs(text: str) -> List[_Token]:
|
|
263
|
+
"""Split ``text`` into box-notation tokens, plus a trailing EOF sentinel."""
|
|
264
|
+
tokens: List[_Token] = []
|
|
265
|
+
i, n = 0, len(text)
|
|
266
|
+
while i < n:
|
|
267
|
+
ch = text[i]
|
|
268
|
+
if ch.isspace():
|
|
269
|
+
i += 1
|
|
270
|
+
continue
|
|
271
|
+
if ch == "-" and i + 1 < n and text[i + 1] == ">":
|
|
272
|
+
tokens.append(("ARROW", "->", i))
|
|
273
|
+
i += 2
|
|
274
|
+
continue
|
|
275
|
+
if ch in "<>":
|
|
276
|
+
if i + 1 < n and text[i + 1] == "=":
|
|
277
|
+
tokens.append(("CMP", ch + "=", i))
|
|
278
|
+
i += 2
|
|
279
|
+
else:
|
|
280
|
+
tokens.append(("CMP", ch, i))
|
|
281
|
+
i += 1
|
|
282
|
+
continue
|
|
283
|
+
if ch.isdigit():
|
|
284
|
+
j = i
|
|
285
|
+
while j < n and text[j].isdigit():
|
|
286
|
+
j += 1
|
|
287
|
+
tokens.append(("NUMBER", text[i:j], i))
|
|
288
|
+
i = j
|
|
289
|
+
continue
|
|
290
|
+
if ch in _GLYPHS:
|
|
291
|
+
tokens.append((_GLYPHS[ch], ch, i))
|
|
292
|
+
i += 1
|
|
293
|
+
continue
|
|
294
|
+
if ch == '"':
|
|
295
|
+
j = i + 1
|
|
296
|
+
while j < n and text[j] != '"':
|
|
297
|
+
j += 1
|
|
298
|
+
if j >= n:
|
|
299
|
+
raise DRSSyntaxError(
|
|
300
|
+
f"parse_drs: unterminated string literal starting at position "
|
|
301
|
+
f"{i} in {text!r}")
|
|
302
|
+
tokens.append(("STRING", text[i + 1:j], i))
|
|
303
|
+
i = j + 1
|
|
304
|
+
continue
|
|
305
|
+
if ch.isalpha():
|
|
306
|
+
j = i
|
|
307
|
+
# `_` is a word CONTINUATION character, never a start: the
|
|
308
|
+
# name validators below decide legality (Has_bond_to is a
|
|
309
|
+
# predicate, c_o1 a constant), the tokenizer only spans the
|
|
310
|
+
# word. Stopping at `_` instead would split `Has_bond_to`
|
|
311
|
+
# into three tokens and mis-parse it.
|
|
312
|
+
while j < n and (text[j].isalnum() or text[j] == "_"):
|
|
313
|
+
j += 1
|
|
314
|
+
word = text[i:j]
|
|
315
|
+
if is_predicate_name(word):
|
|
316
|
+
tokens.append(("PRED", word, i))
|
|
317
|
+
elif is_referent(word):
|
|
318
|
+
tokens.append(("REF", word, i))
|
|
319
|
+
elif is_constant_name(word):
|
|
320
|
+
tokens.append(("CONST", word, i))
|
|
321
|
+
else:
|
|
322
|
+
raise DRSSyntaxError(
|
|
323
|
+
f"parse_drs: {word!r} at position {i} is not a legal referent, "
|
|
324
|
+
f"constant, or predicate name in {text!r}")
|
|
325
|
+
i = j
|
|
326
|
+
continue
|
|
327
|
+
raise DRSSyntaxError(f"parse_drs: unexpected character {ch!r} at position {i} in {text!r}")
|
|
328
|
+
tokens.append(("EOF", "", n))
|
|
329
|
+
return tokens
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
class _DRSParser:
|
|
333
|
+
"""A single parse of one token stream; not re-used across calls."""
|
|
334
|
+
|
|
335
|
+
def __init__(self, tokens: List[_Token], text: str):
|
|
336
|
+
self._tokens = tokens
|
|
337
|
+
self._text = text
|
|
338
|
+
self._i = 0
|
|
339
|
+
self._mapping = NameMapping()
|
|
340
|
+
|
|
341
|
+
def _peek(self, ahead: int = 0) -> _Token:
|
|
342
|
+
return self._tokens[min(self._i + ahead, len(self._tokens) - 1)]
|
|
343
|
+
|
|
344
|
+
def _advance(self) -> _Token:
|
|
345
|
+
tok = self._tokens[self._i]
|
|
346
|
+
self._i += 1
|
|
347
|
+
return tok
|
|
348
|
+
|
|
349
|
+
def _error(self, message: str) -> DRSSyntaxError:
|
|
350
|
+
return DRSSyntaxError(f"parse_drs: {message} in {self._text!r}")
|
|
351
|
+
|
|
352
|
+
def _expect(self, ttype: str, what: str) -> _Token:
|
|
353
|
+
tok = self._peek()
|
|
354
|
+
if tok[0] != ttype:
|
|
355
|
+
found = "end of input" if tok[0] == "EOF" else f"{tok[1]!r}"
|
|
356
|
+
raise self._error(f"expected {what} but found {found} at position {tok[2]}")
|
|
357
|
+
return self._advance()
|
|
358
|
+
|
|
359
|
+
# -- grammar ---------------------------------------------------------- #
|
|
360
|
+
|
|
361
|
+
def parse_document(self) -> DRS:
|
|
362
|
+
result = self._boxexpr()
|
|
363
|
+
eof = self._peek()
|
|
364
|
+
if eof[0] != "EOF":
|
|
365
|
+
raise self._error(
|
|
366
|
+
f"unexpected trailing input {eof[1]!r} at position {eof[2]} (chaining "
|
|
367
|
+
f"multiple ->/∨/~ at the top level is not supported — nest boxes "
|
|
368
|
+
f"explicitly instead)")
|
|
369
|
+
if isinstance(result, DRS):
|
|
370
|
+
return result
|
|
371
|
+
return DRS((), (result,))
|
|
372
|
+
|
|
373
|
+
def _boxexpr(self) -> Union[DRS, Condition]:
|
|
374
|
+
if self._peek()[0] == "TILDE":
|
|
375
|
+
self._advance()
|
|
376
|
+
return Neg(self._box())
|
|
377
|
+
box = self._box()
|
|
378
|
+
nxt = self._peek()
|
|
379
|
+
if nxt[0] == "ARROW":
|
|
380
|
+
self._advance()
|
|
381
|
+
return Impl(box, self._box())
|
|
382
|
+
if nxt[0] == "OR":
|
|
383
|
+
self._advance()
|
|
384
|
+
return Or(box, self._box())
|
|
385
|
+
return box
|
|
386
|
+
|
|
387
|
+
def _box(self) -> DRS:
|
|
388
|
+
self._expect("LB", "'['")
|
|
389
|
+
refs = self._refs()
|
|
390
|
+
self._expect("PIPE", "'|' separating referents from conditions")
|
|
391
|
+
conds = self._conds()
|
|
392
|
+
self._expect("RB", "']'")
|
|
393
|
+
return DRS(tuple(refs), tuple(conds))
|
|
394
|
+
|
|
395
|
+
def _refs(self) -> List[str]:
|
|
396
|
+
if self._peek()[0] != "REF":
|
|
397
|
+
return []
|
|
398
|
+
out = [self._advance()[1]]
|
|
399
|
+
while self._peek()[0] == "COMMA":
|
|
400
|
+
self._advance()
|
|
401
|
+
out.append(self._expect("REF", "a referent name after ','")[1])
|
|
402
|
+
return out
|
|
403
|
+
|
|
404
|
+
def _conds(self) -> List[Condition]:
|
|
405
|
+
if self._peek()[0] == "RB":
|
|
406
|
+
return []
|
|
407
|
+
out = [self._cond()]
|
|
408
|
+
while self._peek()[0] == "COMMA":
|
|
409
|
+
self._advance()
|
|
410
|
+
out.append(self._cond())
|
|
411
|
+
return out
|
|
412
|
+
|
|
413
|
+
def _cond(self) -> Condition:
|
|
414
|
+
t = self._peek()
|
|
415
|
+
if t[0] == "TILDE":
|
|
416
|
+
self._advance()
|
|
417
|
+
return Neg(self._box())
|
|
418
|
+
if t[0] == "LB":
|
|
419
|
+
box = self._box()
|
|
420
|
+
nxt = self._peek()
|
|
421
|
+
if nxt[0] == "ARROW":
|
|
422
|
+
self._advance()
|
|
423
|
+
return Impl(box, self._box())
|
|
424
|
+
if nxt[0] == "OR":
|
|
425
|
+
self._advance()
|
|
426
|
+
return Or(box, self._box())
|
|
427
|
+
raise self._error(
|
|
428
|
+
f"a bare DRS ({box.to_box_notation()}) is not a valid condition by "
|
|
429
|
+
f"itself at position {t[2]} — wrap it in negation (~[...]) or make it "
|
|
430
|
+
f"one side of a duplex condition ([...] -> [...] / [...] ∨ [...])")
|
|
431
|
+
if t[0] == "PRED":
|
|
432
|
+
return self._pred_cond()
|
|
433
|
+
if t[0] in ("REF", "CONST", "STRING"):
|
|
434
|
+
return self._eq_cond()
|
|
435
|
+
raise self._error(
|
|
436
|
+
f"expected a condition (a predicate, an equality, ~[...], or "
|
|
437
|
+
f"[...] -> / ∨ [...]) but found "
|
|
438
|
+
f"{'end of input' if t[0] == 'EOF' else repr(t[1])} at position {t[2]}")
|
|
439
|
+
|
|
440
|
+
def _term(self) -> str:
|
|
441
|
+
t = self._peek()
|
|
442
|
+
if t[0] in ("REF", "CONST"):
|
|
443
|
+
self._advance()
|
|
444
|
+
return t[1]
|
|
445
|
+
if t[0] == "STRING":
|
|
446
|
+
self._advance()
|
|
447
|
+
return self._mapping.for_constant(t[1])
|
|
448
|
+
raise self._error(
|
|
449
|
+
f"expected a referent, constant, or quoted string but found "
|
|
450
|
+
f"{'end of input' if t[0] == 'EOF' else repr(t[1])} at position {t[2]}")
|
|
451
|
+
|
|
452
|
+
def _pred_cond(self) -> Condition:
|
|
453
|
+
name = self._advance()[1]
|
|
454
|
+
self._expect("LP", "'(' after the predicate name")
|
|
455
|
+
args = [self._term()]
|
|
456
|
+
if name == "Card" and self._peek()[0] == "COMMA":
|
|
457
|
+
# Card(g, >=, 3): the second element is a comparison OPERATOR,
|
|
458
|
+
# which no legal term can be -- so peeking one token past the
|
|
459
|
+
# comma decides Card-vs-Pred without ambiguity, and a Pred that
|
|
460
|
+
# happens to be NAMED Card (term-shaped args only) still parses.
|
|
461
|
+
if self._peek(1)[0] in ("CMP", "EQ"):
|
|
462
|
+
self._advance() # the comma
|
|
463
|
+
op = self._advance()[1]
|
|
464
|
+
self._expect("COMMA", "',' before the cardinality bound")
|
|
465
|
+
number = self._expect("NUMBER", "a non-negative integer bound")
|
|
466
|
+
self._expect("RP", "')'")
|
|
467
|
+
return Card(args[0], op, int(number[1]))
|
|
468
|
+
while self._peek()[0] == "COMMA":
|
|
469
|
+
self._advance()
|
|
470
|
+
args.append(self._term())
|
|
471
|
+
self._expect("RP", "')'")
|
|
472
|
+
if name == "Part_of":
|
|
473
|
+
# The typed membership condition (see nodes.Part): exactly two
|
|
474
|
+
# arguments, and a programmatically built Pred("Part_of", (m, g))
|
|
475
|
+
# re-parses as Part -- harmless on purpose, both export to the
|
|
476
|
+
# same Part_of atom.
|
|
477
|
+
if len(args) != 2:
|
|
478
|
+
raise self._error(
|
|
479
|
+
f"Part_of takes exactly two arguments (member, group), "
|
|
480
|
+
f"got {len(args)}")
|
|
481
|
+
return Part(args[0], args[1])
|
|
482
|
+
return Pred(name, tuple(args))
|
|
483
|
+
|
|
484
|
+
def _eq_cond(self) -> Eq:
|
|
485
|
+
a = self._term()
|
|
486
|
+
self._expect("EQ", "'=' (only a predicate application or an equality may "
|
|
487
|
+
"start with a referent/constant/string)")
|
|
488
|
+
b = self._term()
|
|
489
|
+
return Eq(a, b)
|
|
490
|
+
|
|
491
|
+
|
|
492
|
+
def parse_drs(text: str) -> DRS:
|
|
493
|
+
"""Parse ``text`` (the compact box notation documented in the module docstring)
|
|
494
|
+
into a :class:`~unicode_logic_kit.drt.nodes.DRS`. Raises :class:`DRSSyntaxError` on
|
|
495
|
+
malformed input (unbalanced brackets, a bare box used as a condition, chained
|
|
496
|
+
``->``/``∨``/``~``, an illegal referent/constant/predicate token, ...)."""
|
|
497
|
+
parser = _DRSParser(_tokenize_drs(text), text)
|
|
498
|
+
return parser.parse_document()
|
|
499
|
+
|
|
500
|
+
|
|
501
|
+
# =============================================================================
|
|
502
|
+
# 2. SBN subset.
|
|
503
|
+
# =============================================================================
|
|
504
|
+
|
|
505
|
+
_SENSE_RE = re.compile(r"^([a-z][a-z_]*)\.([nvars])\.([0-9]{2})$")
|
|
506
|
+
_ROLE_RE = re.compile(r"^[A-Z][a-zA-Z0-9]*(?:-[A-Z][a-zA-Z0-9]*)?$")
|
|
507
|
+
_OFFSET_RE = re.compile(r"^([+-])([0-9]+)$")
|
|
508
|
+
_ALLCAPS_RE = re.compile(r"^[A-Z_]+$")
|
|
509
|
+
_CONNECTOR_RE = re.compile(r"^([<>])([0-9]+)$")
|
|
510
|
+
_BARE_INTEGER_RE = re.compile(r"^[0-9]+$")
|
|
511
|
+
|
|
512
|
+
_NEGATION = "NEGATION"
|
|
513
|
+
#: The deictic-reference constants Bos (2023) §2.1 documents as bare (unquoted)
|
|
514
|
+
#: literals: the utterance time, its speaker, its addressee, and its location.
|
|
515
|
+
_DEICTIC_CONSTANTS = frozenset({"now", "speaker", "hearer", "here"})
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
@dataclass(frozen=True)
|
|
519
|
+
class SBNMapping:
|
|
520
|
+
"""The mappings :func:`parse_sbn` used, so a lossy sanitization can be undone.
|
|
521
|
+
|
|
522
|
+
``predicates`` maps each original sense token (``"person.n.01"``) to the predicate
|
|
523
|
+
name it became (``"PersonN01"``); ``constants`` maps each original literal constant
|
|
524
|
+
token — a (lower-cased) quoted string, or one of this subset's bare literals (a
|
|
525
|
+
deictic reference or a bare integer, accepted in either dialect — see the module
|
|
526
|
+
docstring's "Bare (unquoted) constants" bullet, part of the shared base grammar) —
|
|
527
|
+
to its sanitized constant name.
|
|
528
|
+
"""
|
|
529
|
+
|
|
530
|
+
predicates: Dict[str, str]
|
|
531
|
+
constants: Dict[str, str]
|
|
532
|
+
|
|
533
|
+
def to_dict(self) -> dict:
|
|
534
|
+
return {"predicates": dict(self.predicates), "constants": dict(self.constants)}
|
|
535
|
+
|
|
536
|
+
|
|
537
|
+
def _sbn_predicate_name(token: str) -> str:
|
|
538
|
+
"""Turn a WordNet-style sense token into a legal predicate name (see the module
|
|
539
|
+
docstring: ``person.n.01`` -> ``PersonN01``). Raises :class:`SBNSyntaxError` if
|
|
540
|
+
``token`` is not a well-formed ``lemma.pos.NN`` sense token."""
|
|
541
|
+
m = _SENSE_RE.fullmatch(token)
|
|
542
|
+
if not m:
|
|
543
|
+
raise SBNSyntaxError(
|
|
544
|
+
f"parse_sbn: sense token {token!r} is not of the form 'lemma.pos.NN' "
|
|
545
|
+
f"(lowercase underscore-joined lemma, pos in n/v/a/r/s, two-digit sense) "
|
|
546
|
+
f"— this SBN subset only supports well-formed WordNet-style sense tokens.")
|
|
547
|
+
lemma, pos, sense = m.groups()
|
|
548
|
+
pascal = "".join(part[:1].upper() + part[1:] for part in lemma.split("_") if part)
|
|
549
|
+
return f"{pascal}{pos.upper()}{sense}"
|
|
550
|
+
|
|
551
|
+
|
|
552
|
+
def _sbn_role_predicate_name(role: str) -> str:
|
|
553
|
+
"""A ROLE (or comparison/temporal/spatial OPERATOR token, e.g. ``EQU`` — lexically
|
|
554
|
+
indistinguishable from a role in this subset, see the module docstring) as a legal
|
|
555
|
+
kit predicate name. Each hyphen-segment of a VerbNet 'Co-' compound (``Co-Theme``,
|
|
556
|
+
:data:`_ROLE_RE`) is already uppercase-initial, so dropping the hyphen keeps it
|
|
557
|
+
PascalCase (``CoTheme``) — the kit PREDICATE convention has no hyphen. A no-op for
|
|
558
|
+
every role that has none, so this changes nothing for an already-accepted input."""
|
|
559
|
+
return role.replace("-", "")
|
|
560
|
+
|
|
561
|
+
|
|
562
|
+
def _split_respecting_quotes(s: str, lineno: int) -> List[str]:
|
|
563
|
+
"""Whitespace-tokenize ``s``, keeping a quoted ``"..."`` span (which may itself
|
|
564
|
+
contain whitespace) as a single token including its quotes."""
|
|
565
|
+
tokens: List[str] = []
|
|
566
|
+
i, n = 0, len(s)
|
|
567
|
+
while i < n:
|
|
568
|
+
while i < n and s[i].isspace():
|
|
569
|
+
i += 1
|
|
570
|
+
if i >= n:
|
|
571
|
+
break
|
|
572
|
+
if s[i] == '"':
|
|
573
|
+
j = i + 1
|
|
574
|
+
while j < n and s[j] != '"':
|
|
575
|
+
j += 1
|
|
576
|
+
if j >= n:
|
|
577
|
+
raise SBNSyntaxError(f"parse_sbn: line {lineno}: unterminated quoted constant")
|
|
578
|
+
tokens.append(s[i:j + 1])
|
|
579
|
+
i = j + 1
|
|
580
|
+
else:
|
|
581
|
+
j = i
|
|
582
|
+
while j < n and not s[j].isspace():
|
|
583
|
+
j += 1
|
|
584
|
+
tokens.append(s[i:j])
|
|
585
|
+
i = j
|
|
586
|
+
return tokens
|
|
587
|
+
|
|
588
|
+
|
|
589
|
+
# Any line-break convention: a document handed over as a string may use CRLF
|
|
590
|
+
# or a bare CR, and a bare-CR one would otherwise be ONE line, its nesting lost.
|
|
591
|
+
_LINE_BREAK_RE = re.compile(r"\r\n?|\n")
|
|
592
|
+
|
|
593
|
+
|
|
594
|
+
def _split_sbn_lines(text: str) -> List[Tuple[str, int, str]]:
|
|
595
|
+
"""Pass 1: strip comments (a PMB ``%%%``-prefixed generation-command header is just
|
|
596
|
+
another ``%...`` comment under this rule) and blank lines. Returns ``(content,
|
|
597
|
+
lineno, leading)`` triples, ``leading`` the RAW leading-whitespace substring —
|
|
598
|
+
whether it means TAB-only nesting depth or is pure cosmetic column alignment
|
|
599
|
+
depends on which of :func:`parse_sbn`'s two dialects the document uses (decided by
|
|
600
|
+
:func:`_uses_connector_dialect` once the whole document has been split), so it is
|
|
601
|
+
NOT validated here."""
|
|
602
|
+
out: List[Tuple[str, int, str]] = []
|
|
603
|
+
for lineno, raw in enumerate(_LINE_BREAK_RE.split(text), start=1):
|
|
604
|
+
line = raw.split("%", 1)[0]
|
|
605
|
+
if not line.strip():
|
|
606
|
+
continue
|
|
607
|
+
no_lead = line.lstrip(" \t")
|
|
608
|
+
leading = line[:len(line) - len(no_lead)]
|
|
609
|
+
out.append((no_lead.strip(), lineno, leading))
|
|
610
|
+
if not out:
|
|
611
|
+
raise SBNSyntaxError("parse_sbn: empty input (no content lines)")
|
|
612
|
+
return out
|
|
613
|
+
|
|
614
|
+
|
|
615
|
+
def _require_tab_indentation(lines: List[Tuple[str, int, str]]) -> None:
|
|
616
|
+
"""The indentation dialect's own rule (see the module docstring): leading
|
|
617
|
+
whitespace is nesting depth and must be TAB-only. Raises on the first line that
|
|
618
|
+
isn't — unchanged from before the connector dialect existed."""
|
|
619
|
+
for _, lineno, leading in lines:
|
|
620
|
+
if " " in leading:
|
|
621
|
+
raise SBNSyntaxError(
|
|
622
|
+
f"parse_sbn: line {lineno} is indented with a space character; this "
|
|
623
|
+
f"SBN subset requires TAB-only indentation")
|
|
624
|
+
|
|
625
|
+
|
|
626
|
+
def _uses_connector_dialect(lines: List[Tuple[str, int, str]]) -> bool:
|
|
627
|
+
"""True iff any box-operator line carries a trailing CONNECTOR token (``<N`` /
|
|
628
|
+
``>N``) — decides which of :func:`parse_sbn`'s two dialects (see the module
|
|
629
|
+
docstring's "Dialect selection") to read the WHOLE document in. None of the
|
|
630
|
+
dialect-1 fixtures in ``tests/test_drt.py`` contain one, so they always resolve
|
|
631
|
+
False here, keeping their parse behaviour exactly as before."""
|
|
632
|
+
for content, _, _ in lines:
|
|
633
|
+
parts = content.split()
|
|
634
|
+
if (parts and _ALLCAPS_RE.fullmatch(parts[0])
|
|
635
|
+
and len(parts) == 2 and _CONNECTOR_RE.fullmatch(parts[1])):
|
|
636
|
+
return True
|
|
637
|
+
return False
|
|
638
|
+
|
|
639
|
+
|
|
640
|
+
@dataclass
|
|
641
|
+
class _SBNLine:
|
|
642
|
+
lineno: int
|
|
643
|
+
depth: int
|
|
644
|
+
index: int
|
|
645
|
+
kind: str # "sense" | "negation"
|
|
646
|
+
predicate: str = "" # sanitized predicate name, kind == "sense" only
|
|
647
|
+
role_targets: tuple = () # tuple of (role, raw_target) pairs
|
|
648
|
+
|
|
649
|
+
|
|
650
|
+
def parse_sbn(text: str) -> Tuple[DRS, SBNMapping]:
|
|
651
|
+
"""Parse ``text`` into a ``(DRS, SBNMapping)`` pair, in whichever of the module
|
|
652
|
+
docstring's two documented SBN dialects ``text`` turns out to use (decided by
|
|
653
|
+
:func:`_uses_connector_dialect`). Raises :class:`SBNSyntaxError` on malformed
|
|
654
|
+
input or on any construct outside the chosen dialect's documented subset (naming
|
|
655
|
+
it explicitly)."""
|
|
656
|
+
raw_lines = _split_sbn_lines(text)
|
|
657
|
+
if _uses_connector_dialect(raw_lines):
|
|
658
|
+
return _parse_sbn_connector(raw_lines)
|
|
659
|
+
_require_tab_indentation(raw_lines)
|
|
660
|
+
return _parse_sbn_classic(raw_lines)
|
|
661
|
+
|
|
662
|
+
|
|
663
|
+
def _parse_sbn_classic(raw_lines: List[Tuple[str, int, str]]) -> Tuple[DRS, SBNMapping]:
|
|
664
|
+
"""Dialect 1, the indentation dialect (see the module docstring): a bare
|
|
665
|
+
``NEGATION`` line opens a sub-box via TAB depth. This is ``parse_sbn``'s ORIGINAL
|
|
666
|
+
subset — every input it already accepted parses to the identical DRS it always
|
|
667
|
+
did; the only additions since (widened ``_ROLE_RE``, the bare deictic/integer
|
|
668
|
+
constants below) are no-ops on any input that does not use them."""
|
|
669
|
+
# Pass 2: tokenize each line's own payload (sense/NEGATION keyword + role/target
|
|
670
|
+
# pairs), without resolving offset targets yet — that needs the full index->kind map.
|
|
671
|
+
pred_mapping: Dict[str, str] = {}
|
|
672
|
+
parsed_lines: List[_SBNLine] = []
|
|
673
|
+
for index, (content, lineno, leading) in enumerate(raw_lines, start=1):
|
|
674
|
+
depth = len(leading)
|
|
675
|
+
parts = _split_respecting_quotes(content, lineno)
|
|
676
|
+
head, rest = parts[0], parts[1:]
|
|
677
|
+
if head == _NEGATION:
|
|
678
|
+
if rest:
|
|
679
|
+
raise SBNSyntaxError(
|
|
680
|
+
f"parse_sbn: line {lineno}: 'NEGATION' takes no roles/targets in "
|
|
681
|
+
f"this subset, found {rest!r}")
|
|
682
|
+
parsed_lines.append(_SBNLine(lineno, depth, index, "negation"))
|
|
683
|
+
continue
|
|
684
|
+
if _ALLCAPS_RE.fullmatch(head):
|
|
685
|
+
raise SBNSyntaxError(
|
|
686
|
+
f"parse_sbn: line {lineno}: box-operator {head!r} is not supported by "
|
|
687
|
+
f"this SBN subset (only NEGATION is) — refusing rather than silently "
|
|
688
|
+
f"misreading it.")
|
|
689
|
+
if len(rest) % 2 != 0:
|
|
690
|
+
raise SBNSyntaxError(
|
|
691
|
+
f"parse_sbn: line {lineno}: role {rest[-1]!r} has no target")
|
|
692
|
+
role_targets = []
|
|
693
|
+
for k in range(0, len(rest), 2):
|
|
694
|
+
role, target = rest[k], rest[k + 1]
|
|
695
|
+
if not _ROLE_RE.fullmatch(role):
|
|
696
|
+
raise SBNSyntaxError(
|
|
697
|
+
f"parse_sbn: line {lineno}: role {role!r} must be uppercase-initial "
|
|
698
|
+
f"alphanumeric, optionally with one hyphenated segment (e.g. "
|
|
699
|
+
f"'Agent', 'Theme', 'Co-Theme')")
|
|
700
|
+
role_targets.append((role, target))
|
|
701
|
+
predicate = pred_mapping.get(head)
|
|
702
|
+
if predicate is None:
|
|
703
|
+
predicate = _sbn_predicate_name(head)
|
|
704
|
+
pred_mapping[head] = predicate
|
|
705
|
+
parsed_lines.append(_SBNLine(lineno, depth, index, "sense", predicate, tuple(role_targets)))
|
|
706
|
+
|
|
707
|
+
n = len(parsed_lines)
|
|
708
|
+
kind_by_index = {pl.index: pl.kind for pl in parsed_lines}
|
|
709
|
+
|
|
710
|
+
# Pass 3: resolve each role's raw target token to a final DRS term string.
|
|
711
|
+
const_mapping = NameMapping()
|
|
712
|
+
resolved: List[Tuple[_SBNLine, tuple]] = []
|
|
713
|
+
for pl in parsed_lines:
|
|
714
|
+
if pl.kind != "sense":
|
|
715
|
+
resolved.append((pl, ()))
|
|
716
|
+
continue
|
|
717
|
+
terms = []
|
|
718
|
+
for role, target in pl.role_targets:
|
|
719
|
+
m = _OFFSET_RE.fullmatch(target)
|
|
720
|
+
if m:
|
|
721
|
+
sign, digits = m.groups()
|
|
722
|
+
delta = int(digits) if sign == "+" else -int(digits)
|
|
723
|
+
target_index = pl.index + delta
|
|
724
|
+
if target_index < 1 or target_index > n:
|
|
725
|
+
raise SBNSyntaxError(
|
|
726
|
+
f"parse_sbn: line {pl.lineno}: offset {target!r} on role "
|
|
727
|
+
f"{role!r} points to line index {target_index}, outside the "
|
|
728
|
+
f"document (1..{n})")
|
|
729
|
+
if kind_by_index[target_index] != "sense":
|
|
730
|
+
raise SBNSyntaxError(
|
|
731
|
+
f"parse_sbn: line {pl.lineno}: offset {target!r} on role "
|
|
732
|
+
f"{role!r} targets line {target_index}, a box-operator line "
|
|
733
|
+
f"with no referent of its own in this subset")
|
|
734
|
+
terms.append((role, f"e{target_index}"))
|
|
735
|
+
elif target.startswith('"') and target.endswith('"') and len(target) >= 2:
|
|
736
|
+
raw_const = target[1:-1].lower()
|
|
737
|
+
terms.append((role, const_mapping.for_constant(raw_const)))
|
|
738
|
+
elif target in _DEICTIC_CONSTANTS or _BARE_INTEGER_RE.fullmatch(target):
|
|
739
|
+
terms.append((role, const_mapping.for_constant(target)))
|
|
740
|
+
else:
|
|
741
|
+
raise SBNSyntaxError(
|
|
742
|
+
f"parse_sbn: line {pl.lineno}: target {target!r} on role {role!r} "
|
|
743
|
+
f"is neither a signed line offset (+N / -N), a quoted constant "
|
|
744
|
+
f'("..."), nor one of this subset\'s bare constant literals '
|
|
745
|
+
f"(now/speaker/hearer/here, or a bare integer).")
|
|
746
|
+
resolved.append((pl, tuple(terms)))
|
|
747
|
+
|
|
748
|
+
# Pass 4: build the nested DRS via an indentation-driven stack of open boxes.
|
|
749
|
+
class _Frame:
|
|
750
|
+
__slots__ = ("depth", "referents", "conditions")
|
|
751
|
+
|
|
752
|
+
def __init__(self, depth: int):
|
|
753
|
+
self.depth = depth
|
|
754
|
+
self.referents: List[str] = []
|
|
755
|
+
self.conditions: List[Condition] = []
|
|
756
|
+
|
|
757
|
+
def _close(frame: "_Frame") -> DRS:
|
|
758
|
+
return DRS(tuple(frame.referents), tuple(frame.conditions))
|
|
759
|
+
|
|
760
|
+
def _pop_negation(stack: List["_Frame"]) -> None:
|
|
761
|
+
closed = stack.pop()
|
|
762
|
+
if not closed.referents and not closed.conditions:
|
|
763
|
+
raise SBNSyntaxError(
|
|
764
|
+
"parse_sbn: a NEGATION line has no content — every NEGATION in this "
|
|
765
|
+
"subset must be followed by at least one more-indented line")
|
|
766
|
+
stack[-1].conditions.append(Neg(_close(closed)))
|
|
767
|
+
|
|
768
|
+
stack: List[_Frame] = [_Frame(0)]
|
|
769
|
+
for pl, terms in resolved:
|
|
770
|
+
d = pl.depth
|
|
771
|
+
while len(stack) > 1 and stack[-1].depth > d:
|
|
772
|
+
_pop_negation(stack)
|
|
773
|
+
current = stack[-1]
|
|
774
|
+
if current.depth != d:
|
|
775
|
+
raise SBNSyntaxError(
|
|
776
|
+
f"parse_sbn: line {pl.lineno}: indentation depth {d} does not open or "
|
|
777
|
+
f"continue any box (expected depth {current.depth})")
|
|
778
|
+
if pl.kind == "negation":
|
|
779
|
+
stack.append(_Frame(d + 1))
|
|
780
|
+
else:
|
|
781
|
+
ref = f"e{pl.index}"
|
|
782
|
+
current.referents.append(ref)
|
|
783
|
+
current.conditions.append(Pred(pl.predicate, (ref,)))
|
|
784
|
+
for role, term in terms:
|
|
785
|
+
current.conditions.append(Pred(_sbn_role_predicate_name(role), (ref, term)))
|
|
786
|
+
|
|
787
|
+
while len(stack) > 1:
|
|
788
|
+
_pop_negation(stack)
|
|
789
|
+
|
|
790
|
+
root = _close(stack[0])
|
|
791
|
+
# Offsets are only range/kind-checked above; an offset that crosses a
|
|
792
|
+
# NEGATION box boundary builds a structurally well-formed but
|
|
793
|
+
# ACCESSIBILITY-violating DRS (review-flagged: a referent used outside
|
|
794
|
+
# the box that declares it). Validate before handing it out so a caller
|
|
795
|
+
# never receives a silently invalid tree — the violation surfaces here,
|
|
796
|
+
# named, instead of downstream in drs_to_fol. nodes.DRS.validate raises a
|
|
797
|
+
# plain ValueError (it has no SBN-specific vocabulary of its own); re-raise
|
|
798
|
+
# as SBNSyntaxError so every parse_sbn refusal is uniformly that one type.
|
|
799
|
+
try:
|
|
800
|
+
root.validate()
|
|
801
|
+
except ValueError as e:
|
|
802
|
+
raise SBNSyntaxError(str(e)) from e
|
|
803
|
+
mapping = SBNMapping(predicates=dict(pred_mapping), constants=dict(const_mapping.constant))
|
|
804
|
+
return root, mapping
|
|
805
|
+
|
|
806
|
+
|
|
807
|
+
def _parse_sbn_connector(raw_lines: List[Tuple[str, int, str]]) -> Tuple[DRS, SBNMapping]:
|
|
808
|
+
"""Dialect 2, the connector dialect (see the module docstring): PMB's own
|
|
809
|
+
released SBN format. Leading whitespace is ignored throughout; box nesting comes
|
|
810
|
+
from each box-operator line's trailing connector (``<N``) rather than
|
|
811
|
+
indentation, and role-hook indices count CONCEPT lines only."""
|
|
812
|
+
|
|
813
|
+
class _Context:
|
|
814
|
+
__slots__ = ("referents", "slots")
|
|
815
|
+
|
|
816
|
+
def __init__(self) -> None:
|
|
817
|
+
self.referents: List[str] = []
|
|
818
|
+
# Each slot is either a ("concept", i) placeholder (i indexes `concepts`
|
|
819
|
+
# below) or a ("neg", child_context_index) placeholder — both resolved
|
|
820
|
+
# into actual Conditions only once every concept's offsets and every
|
|
821
|
+
# child context are known (see the two passes below).
|
|
822
|
+
self.slots: List[Tuple[str, int]] = []
|
|
823
|
+
|
|
824
|
+
contexts: List[_Context] = [_Context()] # context 0 = the outermost, implicit context
|
|
825
|
+
neg_lineno: Dict[int, int] = {} # child context index -> its NEGATION's line
|
|
826
|
+
current = 0
|
|
827
|
+
pred_mapping: Dict[str, str] = {}
|
|
828
|
+
const_mapping = NameMapping()
|
|
829
|
+
concepts: List[dict] = [] # concept-only registry, in document order
|
|
830
|
+
|
|
831
|
+
# Pass 1: one linear scan building the context graph and every concept's raw
|
|
832
|
+
# (unresolved) role targets — an offset may point FORWARD to a concept not yet
|
|
833
|
+
# seen, so resolution is deferred to pass 2 below.
|
|
834
|
+
for content, lineno, _leading in raw_lines:
|
|
835
|
+
head = content.split(None, 1)[0]
|
|
836
|
+
if _ALLCAPS_RE.fullmatch(head):
|
|
837
|
+
rest = content.split()[1:]
|
|
838
|
+
if len(rest) != 1 or not _CONNECTOR_RE.fullmatch(rest[0]):
|
|
839
|
+
raise SBNSyntaxError(
|
|
840
|
+
f"parse_sbn: line {lineno}: a box-operator line in the connector "
|
|
841
|
+
f"dialect (see the module docstring) must be followed by exactly "
|
|
842
|
+
f"one connector (e.g. 'NEGATION <1'), found {rest!r}")
|
|
843
|
+
sign, digits = rest[0][0], rest[0][1:]
|
|
844
|
+
if sign == ">":
|
|
845
|
+
raise SBNSyntaxError(
|
|
846
|
+
f"parse_sbn: line {lineno}: a forward connector ({rest[0]!r}) is "
|
|
847
|
+
f"not supported by this SBN subset — only backward ('<N') "
|
|
848
|
+
f"connectors are.")
|
|
849
|
+
if head != _NEGATION:
|
|
850
|
+
raise SBNSyntaxError(
|
|
851
|
+
f"parse_sbn: line {lineno}: box-operator {head!r} is not "
|
|
852
|
+
f"supported by this SBN subset (only NEGATION is) — refusing "
|
|
853
|
+
f"rather than silently misreading it.")
|
|
854
|
+
n = int(digits)
|
|
855
|
+
new_index = len(contexts)
|
|
856
|
+
target_index = new_index - n
|
|
857
|
+
if n < 1 or target_index < 0:
|
|
858
|
+
raise SBNSyntaxError(
|
|
859
|
+
f"parse_sbn: line {lineno}: connector '<{n}' refers to context "
|
|
860
|
+
f"{target_index}, outside the document (contexts "
|
|
861
|
+
f"0..{new_index - 1} exist at this point)")
|
|
862
|
+
contexts.append(_Context())
|
|
863
|
+
contexts[target_index].slots.append(("neg", new_index))
|
|
864
|
+
neg_lineno[new_index] = lineno
|
|
865
|
+
current = new_index
|
|
866
|
+
continue
|
|
867
|
+
|
|
868
|
+
# A concept (sense) line — tokenized quote-aware (unlike the box-operator
|
|
869
|
+
# check above, a concept's role targets may be quoted strings).
|
|
870
|
+
toks = _split_respecting_quotes(content, lineno)
|
|
871
|
+
head, rest = toks[0], toks[1:]
|
|
872
|
+
if len(rest) % 2 != 0:
|
|
873
|
+
raise SBNSyntaxError(
|
|
874
|
+
f"parse_sbn: line {lineno}: role {rest[-1]!r} has no target")
|
|
875
|
+
role_targets = []
|
|
876
|
+
for k in range(0, len(rest), 2):
|
|
877
|
+
role, target = rest[k], rest[k + 1]
|
|
878
|
+
if not _ROLE_RE.fullmatch(role):
|
|
879
|
+
raise SBNSyntaxError(
|
|
880
|
+
f"parse_sbn: line {lineno}: role {role!r} must be uppercase-initial "
|
|
881
|
+
f"alphanumeric, optionally with one hyphenated segment (e.g. "
|
|
882
|
+
f"'Agent', 'Theme', 'Co-Theme')")
|
|
883
|
+
role_targets.append((role, target))
|
|
884
|
+
predicate = pred_mapping.get(head)
|
|
885
|
+
if predicate is None:
|
|
886
|
+
predicate = _sbn_predicate_name(head)
|
|
887
|
+
pred_mapping[head] = predicate
|
|
888
|
+
concept_index = len(concepts) # 0-based position
|
|
889
|
+
ref = f"e{concept_index + 1}"
|
|
890
|
+
contexts[current].referents.append(ref)
|
|
891
|
+
contexts[current].slots.append(("concept", concept_index))
|
|
892
|
+
concepts.append({"lineno": lineno, "ref": ref, "predicate": predicate,
|
|
893
|
+
"role_targets": role_targets, "resolved": None})
|
|
894
|
+
|
|
895
|
+
n_concepts = len(concepts)
|
|
896
|
+
|
|
897
|
+
# Pass 2: resolve every role-hook target now that each concept's CONCEPT-ONLY
|
|
898
|
+
# position (1-based, box-operator lines excluded — see the module docstring) is
|
|
899
|
+
# known.
|
|
900
|
+
for i, c in enumerate(concepts, start=1):
|
|
901
|
+
resolved_terms = []
|
|
902
|
+
for role, target in c["role_targets"]:
|
|
903
|
+
m = _OFFSET_RE.fullmatch(target)
|
|
904
|
+
if m:
|
|
905
|
+
sign, digits = m.groups()
|
|
906
|
+
delta = int(digits) if sign == "+" else -int(digits)
|
|
907
|
+
target_index = i + delta
|
|
908
|
+
if target_index < 1 or target_index > n_concepts:
|
|
909
|
+
raise SBNSyntaxError(
|
|
910
|
+
f"parse_sbn: line {c['lineno']}: offset {target!r} on role "
|
|
911
|
+
f"{role!r} points to concept index {target_index}, outside "
|
|
912
|
+
f"the document ({n_concepts} concept(s))")
|
|
913
|
+
resolved_terms.append((role, concepts[target_index - 1]["ref"]))
|
|
914
|
+
elif target.startswith('"') and target.endswith('"') and len(target) >= 2:
|
|
915
|
+
raw_const = target[1:-1].lower()
|
|
916
|
+
resolved_terms.append((role, const_mapping.for_constant(raw_const)))
|
|
917
|
+
elif target in _DEICTIC_CONSTANTS or _BARE_INTEGER_RE.fullmatch(target):
|
|
918
|
+
resolved_terms.append((role, const_mapping.for_constant(target)))
|
|
919
|
+
elif _CONNECTOR_RE.fullmatch(target):
|
|
920
|
+
raise SBNSyntaxError(
|
|
921
|
+
f"parse_sbn: line {c['lineno']}: role {role!r}'s target "
|
|
922
|
+
f"{target!r} is a context/box reference (an embedded-clause or "
|
|
923
|
+
f"propositional-attitude argument) — this SBN subset does not "
|
|
924
|
+
f"support box-valued role targets.")
|
|
925
|
+
else:
|
|
926
|
+
raise SBNSyntaxError(
|
|
927
|
+
f"parse_sbn: line {c['lineno']}: target {target!r} on role "
|
|
928
|
+
f"{role!r} is neither a signed concept offset (+N / -N), a "
|
|
929
|
+
f'quoted constant ("..."), nor one of this subset\'s bare '
|
|
930
|
+
f"constant literals (now/speaker/hearer/here, or a bare integer).")
|
|
931
|
+
c["resolved"] = resolved_terms
|
|
932
|
+
|
|
933
|
+
# Pass 3: close every context bottom-up. A "neg" slot always names a context
|
|
934
|
+
# with a STRICTLY GREATER index than its own (target_index < new_index above),
|
|
935
|
+
# so resolving from the highest index down guarantees a child is already closed
|
|
936
|
+
# by the time its parent needs it.
|
|
937
|
+
closed: Dict[int, DRS] = {}
|
|
938
|
+
for k in range(len(contexts) - 1, -1, -1):
|
|
939
|
+
conditions: List[Condition] = []
|
|
940
|
+
for kind, idx in contexts[k].slots:
|
|
941
|
+
if kind == "concept":
|
|
942
|
+
c = concepts[idx]
|
|
943
|
+
conditions.append(Pred(c["predicate"], (c["ref"],)))
|
|
944
|
+
for role, term in c["resolved"]:
|
|
945
|
+
conditions.append(Pred(_sbn_role_predicate_name(role), (c["ref"], term)))
|
|
946
|
+
else: # "neg"
|
|
947
|
+
conditions.append(Neg(closed[idx]))
|
|
948
|
+
if k != 0 and not contexts[k].referents and not conditions:
|
|
949
|
+
raise SBNSyntaxError(
|
|
950
|
+
f"parse_sbn: line {neg_lineno[k]}: NEGATION introduces an empty "
|
|
951
|
+
f"context — every NEGATION in this subset must be followed by at "
|
|
952
|
+
f"least one concept before the next box-operator or the end of the "
|
|
953
|
+
f"document")
|
|
954
|
+
closed[k] = DRS(tuple(contexts[k].referents), tuple(conditions))
|
|
955
|
+
|
|
956
|
+
root = closed[0]
|
|
957
|
+
# See the identical comment in _parse_sbn_classic above: validated here rather
|
|
958
|
+
# than left for drs_to_fol to discover downstream, and re-raised as
|
|
959
|
+
# SBNSyntaxError so every parse_sbn refusal is uniformly that one type.
|
|
960
|
+
try:
|
|
961
|
+
root.validate()
|
|
962
|
+
except ValueError as e:
|
|
963
|
+
raise SBNSyntaxError(str(e)) from e
|
|
964
|
+
mapping = SBNMapping(predicates=dict(pred_mapping), constants=dict(const_mapping.constant))
|
|
965
|
+
return root, mapping
|