deontic 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- deontic/__init__.py +32 -0
- deontic/dictionary.py +203 -0
- deontic/errors.py +96 -0
- deontic/hooks.py +100 -0
- deontic/lexer.py +194 -0
- deontic/parser.py +744 -0
- deontic/resolve.py +261 -0
- deontic-0.1.0.dist-info/METADATA +86 -0
- deontic-0.1.0.dist-info/RECORD +12 -0
- deontic-0.1.0.dist-info/WHEEL +5 -0
- deontic-0.1.0.dist-info/licenses/LICENSE +21 -0
- deontic-0.1.0.dist-info/top_level.txt +1 -0
deontic/__init__.py
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""The deontic constraint language.
|
|
2
|
+
|
|
3
|
+
`parse` turns one sentence into the AST the conformance corpus fixes
|
|
4
|
+
(conformance/ast.schema.json); `load` reads a dictionary; `resolve` checks a
|
|
5
|
+
parsed sentence against it. `deontic.hooks` is the contract between a
|
|
6
|
+
lexicon's code half and any implementation. No evaluator lives here.
|
|
7
|
+
|
|
8
|
+
The version is read from installed metadata rather than written as a literal,
|
|
9
|
+
so the value in a published wheel and the value this module reports cannot
|
|
10
|
+
disagree.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from importlib.metadata import PackageNotFoundError, version as _version
|
|
14
|
+
|
|
15
|
+
try:
|
|
16
|
+
__version__ = _version("deontic")
|
|
17
|
+
except PackageNotFoundError: # running from a source tree, not installed
|
|
18
|
+
__version__ = "0.0.0+unknown"
|
|
19
|
+
|
|
20
|
+
from .dictionary import Dictionary, DictionaryError, load # noqa: E402
|
|
21
|
+
from .errors import DeonticError, LexicalError, ResolutionError # noqa: E402
|
|
22
|
+
from .parser import parse # noqa: E402
|
|
23
|
+
from .resolve import Scope, resolve # noqa: E402
|
|
24
|
+
from .hooks import ( # noqa: E402
|
|
25
|
+
ENTRY_POINT_GROUP, Attestation, Context, Entity, Hook, Lexicon, Relation, Witness, World,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
__all__ = [
|
|
29
|
+
"__version__", "parse", "resolve", "load", "Dictionary", "DictionaryError", "Scope",
|
|
30
|
+
"DeonticError", "LexicalError", "ResolutionError", "ENTRY_POINT_GROUP", "Attestation", "Context", "Entity", "Hook",
|
|
31
|
+
"Lexicon", "Relation", "Witness", "World",
|
|
32
|
+
]
|
deontic/dictionary.py
ADDED
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
"""Loading a dictionary (conformance/dictionary.schema.json) into a form the
|
|
2
|
+
resolver can query: every declared name with its role, field kinds, tag
|
|
3
|
+
sets, relations by possessive word, verbs with their patterns.
|
|
4
|
+
|
|
5
|
+
Loading checks what the JSON Schema cannot: that references point at
|
|
6
|
+
declared types, that a pattern names real fields of the right kind, and
|
|
7
|
+
that no name is declared in two roles.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
from dataclasses import dataclass, field
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
SCALARS = ("text", "number", "date", "boolean")
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class DictionaryError(ValueError):
|
|
21
|
+
pass
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def field_kind(f: Any) -> str:
|
|
25
|
+
return f if isinstance(f, str) else f["kind"]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass
|
|
29
|
+
class Dictionary:
|
|
30
|
+
data: dict
|
|
31
|
+
types: dict[str, dict] = field(default_factory=dict)
|
|
32
|
+
roles: dict[str, str] = field(default_factory=dict) # name (lower) → type | plural | event | cadence | metric | attester | attester_plural | term | reference
|
|
33
|
+
canonical: dict[str, str] = field(default_factory=dict) # name (lower) → declared spelling
|
|
34
|
+
plural_of: dict[str, str] = field(default_factory=dict) # plural (lower) → singular type
|
|
35
|
+
verbs: dict[str, dict] = field(default_factory=dict) # canonical verb → declaration
|
|
36
|
+
verb_of: dict[str, str] = field(default_factory=dict) # any spelling → canonical
|
|
37
|
+
relations: dict[str, dict] = field(default_factory=dict)
|
|
38
|
+
|
|
39
|
+
@property
|
|
40
|
+
def name(self) -> str:
|
|
41
|
+
return self.data["name"]
|
|
42
|
+
|
|
43
|
+
@property
|
|
44
|
+
def strictness(self) -> str:
|
|
45
|
+
return self.data.get("strictness", "permissive")
|
|
46
|
+
|
|
47
|
+
# --- queries ---------------------------------------------------------------------
|
|
48
|
+
|
|
49
|
+
def role(self, name: str) -> str | None:
|
|
50
|
+
return self.roles.get(name.lower())
|
|
51
|
+
|
|
52
|
+
def type_of(self, name: str) -> str | None:
|
|
53
|
+
"""The type a marked name denotes: itself, or the singular of a plural."""
|
|
54
|
+
r = self.role(name)
|
|
55
|
+
if r == "type":
|
|
56
|
+
return self.canonical[name.lower()]
|
|
57
|
+
if r == "plural":
|
|
58
|
+
return self.plural_of[name.lower()]
|
|
59
|
+
return None
|
|
60
|
+
|
|
61
|
+
def fields(self, type_name: str) -> dict:
|
|
62
|
+
return self.types[type_name].get("fields", {})
|
|
63
|
+
|
|
64
|
+
def tags(self, type_name: str) -> set[str]:
|
|
65
|
+
return {t if isinstance(t, str) else t["name"] for t in self.types[type_name].get("tags", [])}
|
|
66
|
+
|
|
67
|
+
def step(self, type_name: str, word: str) -> tuple[str | None, str | None]:
|
|
68
|
+
"""Follow one possessive step. Returns (target type, error code or None)."""
|
|
69
|
+
f = self.fields(type_name).get(word)
|
|
70
|
+
if f is not None:
|
|
71
|
+
if field_kind(f) == "reference":
|
|
72
|
+
return f["to"], None
|
|
73
|
+
return None, "not_a_reference"
|
|
74
|
+
for r in self.relations.values():
|
|
75
|
+
if type_name in r["from"] and r.get("as", r["to"]) == word:
|
|
76
|
+
return r["to"], None
|
|
77
|
+
return None, "unknown_field"
|
|
78
|
+
|
|
79
|
+
def verb(self, spelling: str) -> dict | None:
|
|
80
|
+
c = self.verb_of.get(spelling.lower())
|
|
81
|
+
return self.verbs[c] if c else None
|
|
82
|
+
|
|
83
|
+
def verb_dated(self, decl: dict) -> bool:
|
|
84
|
+
if "pattern" in decl:
|
|
85
|
+
return "at" in decl["pattern"]
|
|
86
|
+
return bool(decl.get("dated", False))
|
|
87
|
+
|
|
88
|
+
def suggest(self, type_name: str, word: str) -> str | None:
|
|
89
|
+
import difflib
|
|
90
|
+
names = list(self.fields(type_name))
|
|
91
|
+
m = difflib.get_close_matches(word, names, n=1, cutoff=0.75)
|
|
92
|
+
return m[0] if m else None
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _declare(d: Dictionary, name: str, role: str, canonical: str | None = None):
|
|
96
|
+
key = name.lower()
|
|
97
|
+
if key in d.roles and d.roles[key] != role:
|
|
98
|
+
raise DictionaryError(f"{name!r} declared as both {d.roles[key]} and {role}")
|
|
99
|
+
d.roles[key] = role
|
|
100
|
+
d.canonical[key] = canonical or name
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def load(source: dict | str | Path) -> Dictionary:
|
|
104
|
+
data = source if isinstance(source, dict) else json.loads(Path(source).read_text())
|
|
105
|
+
d = Dictionary(data=data)
|
|
106
|
+
d.types = data.get("types", {})
|
|
107
|
+
for name, t in d.types.items():
|
|
108
|
+
_declare(d, name, "type")
|
|
109
|
+
plural = t.get("plural", name + "s")
|
|
110
|
+
_declare(d, plural, "plural")
|
|
111
|
+
d.plural_of[plural.lower()] = name
|
|
112
|
+
for fname, f in t.get("fields", {}).items():
|
|
113
|
+
k = field_kind(f)
|
|
114
|
+
if k == "reference":
|
|
115
|
+
if f["to"] not in d.types:
|
|
116
|
+
raise DictionaryError(f"{name}.{fname} refers to undeclared type {f['to']!r}")
|
|
117
|
+
elif k == "list":
|
|
118
|
+
if f["of"] not in SCALARS:
|
|
119
|
+
raise DictionaryError(f"{name}.{fname}: list of {f['of']!r}")
|
|
120
|
+
elif k not in SCALARS:
|
|
121
|
+
raise DictionaryError(f"{name}.{fname}: unknown kind {k!r}")
|
|
122
|
+
for name, r in data.get("relations", {}).items():
|
|
123
|
+
for s in r["from"]:
|
|
124
|
+
if s not in d.types:
|
|
125
|
+
raise DictionaryError(f"relation {name}: undeclared subject type {s!r}")
|
|
126
|
+
if r["to"] not in d.types:
|
|
127
|
+
raise DictionaryError(f"relation {name}: undeclared object type {r['to']!r}")
|
|
128
|
+
d.relations[name] = r
|
|
129
|
+
for key, role in (("events", "event"), ("cadences", "cadence"), ("metrics", "metric"), ("attesters", "attester")):
|
|
130
|
+
for name, decl in data.get(key, {}).items():
|
|
131
|
+
_declare(d, name, role)
|
|
132
|
+
if role == "attester" and "plural" in decl:
|
|
133
|
+
_declare(d, decl["plural"], "attester_plural", canonical=name)
|
|
134
|
+
if role == "event":
|
|
135
|
+
if decl["type"] not in d.types or field_kind(d.fields(decl["type"]).get(decl["field"], "text")) != "date":
|
|
136
|
+
raise DictionaryError(f"event {name}: not a date field of a declared type")
|
|
137
|
+
if role == "metric" and decl["of"] not in d.types:
|
|
138
|
+
raise DictionaryError(f"metric {name}: undeclared type {decl['of']!r}")
|
|
139
|
+
for name, v in data.get("verbs", {}).items():
|
|
140
|
+
d.verbs[name] = v
|
|
141
|
+
d.verb_of[name.lower()] = name
|
|
142
|
+
for syn in v.get("synonyms", []):
|
|
143
|
+
d.verb_of[syn.lower()] = name
|
|
144
|
+
_check_pattern(d, name, v)
|
|
145
|
+
for sentence in data.get("definitions", []):
|
|
146
|
+
from .lexer import fragments
|
|
147
|
+
first = next(f for f in fragments(sentence) if f.kind == "term")
|
|
148
|
+
_declare(d, first.text, "reference" if " refers to " in sentence else "term")
|
|
149
|
+
return d
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _check_pattern(d: Dictionary, name: str, v: dict):
|
|
153
|
+
p = v.get("pattern")
|
|
154
|
+
if p is None:
|
|
155
|
+
if "dated" not in v:
|
|
156
|
+
raise DictionaryError(f"verb {name}: a hook verb must declare dated")
|
|
157
|
+
return
|
|
158
|
+
if "dated" in v:
|
|
159
|
+
raise DictionaryError(f"verb {name}: a verb is data or code, never both")
|
|
160
|
+
subj, obj = v.get("subject"), v.get("object", "entity")
|
|
161
|
+
|
|
162
|
+
def ref(type_name, fname, to):
|
|
163
|
+
f = d.fields(type_name).get(fname)
|
|
164
|
+
if f is None or field_kind(f) != "reference" or (to and f["to"] != to):
|
|
165
|
+
raise DictionaryError(f"verb {name}: {type_name}.{fname} is not a reference to {to}")
|
|
166
|
+
|
|
167
|
+
def date(type_name, fname):
|
|
168
|
+
f = d.fields(type_name).get(fname)
|
|
169
|
+
if f is None or field_kind(f) != "date":
|
|
170
|
+
raise DictionaryError(f"verb {name}: {type_name}.{fname} is not a date")
|
|
171
|
+
|
|
172
|
+
if p["kind"] == "field":
|
|
173
|
+
if subj is None:
|
|
174
|
+
raise DictionaryError(f"verb {name}: a field pattern needs a subject type")
|
|
175
|
+
if "object" in p:
|
|
176
|
+
ref(subj, p["object"], obj)
|
|
177
|
+
if "at" in p:
|
|
178
|
+
date(subj, p["at"])
|
|
179
|
+
elif p["kind"] == "record":
|
|
180
|
+
if p["via"] not in d.types:
|
|
181
|
+
raise DictionaryError(f"verb {name}: via type {p['via']!r} undeclared")
|
|
182
|
+
ref(p["via"], p["subject"], subj)
|
|
183
|
+
if p.get("object") == "self":
|
|
184
|
+
if obj != p["via"]:
|
|
185
|
+
raise DictionaryError(f"verb {name}: object self means the object type is {p['via']}")
|
|
186
|
+
elif "object" in p:
|
|
187
|
+
ref(p["via"], p["object"], obj if obj not in ("entity", "none") else None)
|
|
188
|
+
if "at" in p:
|
|
189
|
+
date(p["via"], p["at"])
|
|
190
|
+
elif p["kind"] == "relation":
|
|
191
|
+
r = d.relations.get(p["name"])
|
|
192
|
+
if r is None:
|
|
193
|
+
raise DictionaryError(f"verb {name}: relation {p['name']!r} undeclared")
|
|
194
|
+
if subj and subj not in r["from"]:
|
|
195
|
+
raise DictionaryError(f"verb {name}: subject {subj} is not a subject of {p['name']}")
|
|
196
|
+
if obj not in ("entity", "none") and obj != r["to"]:
|
|
197
|
+
raise DictionaryError(f"verb {name}: object {obj} is not the object of {p['name']}")
|
|
198
|
+
elif p["kind"] == "reference":
|
|
199
|
+
if obj != "entity":
|
|
200
|
+
raise DictionaryError(f"verb {name}: a reference pattern takes any entity")
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
__all__ = ["Dictionary", "DictionaryError", "load", "field_kind"]
|
deontic/errors.py
ADDED
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
"""Errors, named by the codes the conformance corpus uses."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class DeonticError(Exception):
|
|
7
|
+
code = "error"
|
|
8
|
+
|
|
9
|
+
def __init__(self, message: str = "", **details):
|
|
10
|
+
super().__init__(message or self.code)
|
|
11
|
+
self.message = message or self.code
|
|
12
|
+
self.details = details
|
|
13
|
+
|
|
14
|
+
def as_dict(self) -> dict:
|
|
15
|
+
return {"code": self.code, **{k: v for k, v in self.details.items() if v is not None}}
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class LexicalError(DeonticError):
|
|
19
|
+
"""A sentence that cannot be read at all, in either strictness setting."""
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class UnmatchedMarker(LexicalError):
|
|
23
|
+
code = "unmatched_marker"
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class EmptyMarker(LexicalError):
|
|
27
|
+
code = "empty_marker"
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class SigilInsideMarker(LexicalError):
|
|
31
|
+
code = "sigil_inside_marker"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class UnterminatedString(LexicalError):
|
|
35
|
+
code = "unterminated_string"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class UnbalancedParenthesis(LexicalError):
|
|
39
|
+
code = "unbalanced_parenthesis"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class MissingPeriod(LexicalError):
|
|
43
|
+
code = "missing_period"
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class NoShape(DeonticError):
|
|
47
|
+
"""Strict only: marked text matching no shape."""
|
|
48
|
+
code = "no_shape"
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class UnresolvedAnaphora(DeonticError):
|
|
52
|
+
code = "unresolved_anaphora"
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class ResolutionError(DeonticError):
|
|
56
|
+
"""An error found against a dictionary."""
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class UnknownTerm(ResolutionError):
|
|
60
|
+
code = "unknown_term"
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class UnknownField(ResolutionError):
|
|
64
|
+
code = "unknown_field"
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class NotAReference(ResolutionError):
|
|
68
|
+
code = "not_a_reference"
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
class UnknownTag(ResolutionError):
|
|
72
|
+
code = "unknown_tag"
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class TypeMismatch(ResolutionError):
|
|
76
|
+
code = "type_mismatch"
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
class UnknownVerb(ResolutionError):
|
|
80
|
+
code = "unknown_verb"
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
class ObjectKindMismatch(ResolutionError):
|
|
84
|
+
code = "object_kind_mismatch"
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class UnknownRubric(ResolutionError):
|
|
88
|
+
code = "unknown_rubric"
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
class ConfidenceRequiresSystem(ResolutionError):
|
|
92
|
+
code = "confidence_requires_system"
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
class UndatedVerb(ResolutionError):
|
|
96
|
+
code = "undated_verb"
|
deontic/hooks.py
ADDED
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
"""The Python form of a lexicon's code half (docs/hooks.md §4).
|
|
2
|
+
|
|
3
|
+
A verb the dictionary declares without a `pattern` is a hook verb. A lexicon
|
|
4
|
+
may ship a callable for it; an implementation that can run the callable
|
|
5
|
+
evaluates the verb, one that cannot skips constraints using it. This module
|
|
6
|
+
is the whole contract: the witness record, the read-only world and context
|
|
7
|
+
a hook receives, and the entry-point group lexicons register under.
|
|
8
|
+
|
|
9
|
+
Stdlib only, and no evaluator lives here.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from dataclasses import dataclass, field
|
|
15
|
+
from datetime import date
|
|
16
|
+
from typing import Any, Callable, Iterable, Mapping, Protocol, Sequence
|
|
17
|
+
|
|
18
|
+
ENTRY_POINT_GROUP = "deontic.lexicons"
|
|
19
|
+
"""Python lexicons register here, named after their dictionary. The entry
|
|
20
|
+
point resolves to an object with a `hooks` mapping from verb name to Hook."""
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass(frozen=True)
|
|
24
|
+
class Entity:
|
|
25
|
+
id: str
|
|
26
|
+
type: str
|
|
27
|
+
tags: frozenset[str] = frozenset()
|
|
28
|
+
fields: Mapping[str, Any] = field(default_factory=dict)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass(frozen=True)
|
|
32
|
+
class Relation:
|
|
33
|
+
name: str
|
|
34
|
+
subject: str
|
|
35
|
+
object: str
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass(frozen=True)
|
|
39
|
+
class Attestation:
|
|
40
|
+
subject: str
|
|
41
|
+
attester: str
|
|
42
|
+
claim: str
|
|
43
|
+
date: date
|
|
44
|
+
confidence: float | None = None
|
|
45
|
+
system: str | None = None
|
|
46
|
+
attester_id: str | None = None
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@dataclass(frozen=True)
|
|
50
|
+
class Witness:
|
|
51
|
+
"""One piece of evidence that a verb holds of a subject.
|
|
52
|
+
|
|
53
|
+
`object` is the entity the verb relates the subject to, if the verb takes
|
|
54
|
+
one; `at` is when, if the verb is dated; `via` is the record that carries
|
|
55
|
+
the evidence when it is neither the subject nor the object.
|
|
56
|
+
"""
|
|
57
|
+
object: str | None = None
|
|
58
|
+
at: date | None = None
|
|
59
|
+
via: str | None = None
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class World(Protocol):
|
|
63
|
+
"""Read access to the closed world, and nothing outside it."""
|
|
64
|
+
|
|
65
|
+
def entity(self, id: str) -> Entity | None: ...
|
|
66
|
+
def entities(self, type: str) -> Iterable[Entity]: ...
|
|
67
|
+
def relations(self, name: str) -> Iterable[Relation]: ...
|
|
68
|
+
def attestations(self, subject: str) -> Iterable[Attestation]: ...
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
@dataclass(frozen=True)
|
|
72
|
+
class Context:
|
|
73
|
+
as_of: date
|
|
74
|
+
parameters: Mapping[str, Any]
|
|
75
|
+
verb: Mapping[str, Any]
|
|
76
|
+
"""The verb's declaration from the dictionary, as loaded."""
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
Hook = Callable[[Entity, World, Context], Iterable[Witness]]
|
|
80
|
+
"""A hook is pure: the same subject, world and context give the same
|
|
81
|
+
witnesses. It sees no sentence, filter or time expression."""
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
class Lexicon(Protocol):
|
|
85
|
+
"""What an entry point in ENTRY_POINT_GROUP resolves to."""
|
|
86
|
+
|
|
87
|
+
hooks: Mapping[str, Hook]
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def dated(hook: Hook) -> bool:
|
|
91
|
+
"""Whether a hook's witnesses carry a date, per its `dated` attribute; a
|
|
92
|
+
hook without one is undated, so a time expression on its verb is an
|
|
93
|
+
`undated_verb` error rather than a silent pass."""
|
|
94
|
+
return bool(getattr(hook, "dated", False))
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
__all__ = [
|
|
98
|
+
"ENTRY_POINT_GROUP", "Entity", "Relation", "Attestation", "Witness",
|
|
99
|
+
"World", "Context", "Hook", "Lexicon", "dated",
|
|
100
|
+
]
|
deontic/lexer.py
ADDED
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
"""Marker lexing and word tokenising (design §10, corpus README).
|
|
2
|
+
|
|
3
|
+
Two passes. The first splits the source into typed fragments: `$Term$`,
|
|
4
|
+
`$$Terms$$`, `@verb@`, `#90 days#`, `<parameter>` and the prose between
|
|
5
|
+
them. Only editors write markers, so any unmatched or empty marker is an
|
|
6
|
+
error. The second turns the prose into words, numbers, dates, strings and
|
|
7
|
+
punctuation, giving the parser one flat token stream.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import re
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
from typing import Any
|
|
15
|
+
|
|
16
|
+
from .errors import (
|
|
17
|
+
EmptyMarker, MissingPeriod, SigilInsideMarker, UnbalancedParenthesis,
|
|
18
|
+
UnmatchedMarker, UnterminatedString,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
SIGILS = "$@#<>"
|
|
22
|
+
UNITS = {
|
|
23
|
+
"hour": "hour", "hours": "hour", "day": "day", "days": "day", "week": "week", "weeks": "week",
|
|
24
|
+
"month": "month", "months": "month", "quarter": "quarter", "quarters": "quarter",
|
|
25
|
+
"year": "year", "years": "year", "percent": "percent",
|
|
26
|
+
}
|
|
27
|
+
CALENDAR_UNITS = {"calendar week": "calendar week", "calendar weeks": "calendar week",
|
|
28
|
+
"calendar month": "calendar month", "calendar months": "calendar month",
|
|
29
|
+
"calendar year": "calendar year", "calendar years": "calendar year"}
|
|
30
|
+
NUMBER_WORDS = {"one": 1, "two": 2, "three": 3, "four": 4, "five": 5, "six": 6, "seven": 7,
|
|
31
|
+
"eight": 8, "nine": 9, "ten": 10}
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass(frozen=True)
|
|
35
|
+
class Fragment:
|
|
36
|
+
kind: str # prose | term | verb | quantity | parameter
|
|
37
|
+
text: str
|
|
38
|
+
pos: int
|
|
39
|
+
plural: bool = False
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass(frozen=True)
|
|
43
|
+
class Token:
|
|
44
|
+
kind: str # word | punct | string | number | date | term | verb | quantity | parameter
|
|
45
|
+
value: Any
|
|
46
|
+
pos: int
|
|
47
|
+
plural: bool = False
|
|
48
|
+
raw: str = ""
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def normalise(text: str) -> str:
|
|
52
|
+
return " ".join(text.split())
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def quantity(text: str) -> dict:
|
|
56
|
+
"""`90 days` → {value: 90, unit: day}; `7` → {value: 7}; `year` → {value: 1, unit: year}."""
|
|
57
|
+
words = normalise(text).lower().split()
|
|
58
|
+
if not words:
|
|
59
|
+
raise ValueError("empty quantity")
|
|
60
|
+
value = None
|
|
61
|
+
if re.fullmatch(r"\d+(\.\d+)?", words[0]):
|
|
62
|
+
value = float(words[0]) if "." in words[0] else int(words[0])
|
|
63
|
+
words = words[1:]
|
|
64
|
+
elif words[0] in NUMBER_WORDS:
|
|
65
|
+
value = NUMBER_WORDS[words[0]]
|
|
66
|
+
words = words[1:]
|
|
67
|
+
unit = " ".join(words)
|
|
68
|
+
if unit in CALENDAR_UNITS:
|
|
69
|
+
unit = CALENDAR_UNITS[unit]
|
|
70
|
+
elif unit in UNITS:
|
|
71
|
+
unit = UNITS[unit]
|
|
72
|
+
elif unit:
|
|
73
|
+
raise ValueError(f"unknown unit {unit!r}")
|
|
74
|
+
if value is None:
|
|
75
|
+
if not unit:
|
|
76
|
+
raise ValueError("quantity without a value")
|
|
77
|
+
value = 1
|
|
78
|
+
out: dict = {"value": value}
|
|
79
|
+
if unit:
|
|
80
|
+
out["unit"] = unit
|
|
81
|
+
return out
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def fragments(source: str) -> list[Fragment]:
|
|
85
|
+
out: list[Fragment] = []
|
|
86
|
+
i, n, prose_start = 0, len(source), 0
|
|
87
|
+
while i < n:
|
|
88
|
+
c = source[i]
|
|
89
|
+
if c not in "$@#<":
|
|
90
|
+
i += 1
|
|
91
|
+
continue
|
|
92
|
+
if prose_start < i:
|
|
93
|
+
out.append(Fragment("prose", source[prose_start:i], prose_start))
|
|
94
|
+
if c == "$" and source.startswith("$$", i):
|
|
95
|
+
open_m, close_m, kind, plural = "$$", "$$", "term", True
|
|
96
|
+
else:
|
|
97
|
+
open_m = c
|
|
98
|
+
close_m = {"$": "$", "@": "@", "#": "#", "<": ">"}[c]
|
|
99
|
+
kind = {"$": "term", "@": "verb", "#": "quantity", "<": "parameter"}[c]
|
|
100
|
+
plural = False
|
|
101
|
+
start = i + len(open_m)
|
|
102
|
+
end = source.find(close_m, start)
|
|
103
|
+
if end < 0:
|
|
104
|
+
raise UnmatchedMarker(f"unmatched {open_m!r}", position=i)
|
|
105
|
+
content = source[start:end]
|
|
106
|
+
if any(s in content for s in SIGILS):
|
|
107
|
+
raise SigilInsideMarker("markers do not nest", position=i)
|
|
108
|
+
content = normalise(content)
|
|
109
|
+
if not content:
|
|
110
|
+
raise EmptyMarker("empty marker", position=i)
|
|
111
|
+
out.append(Fragment(kind, content, i, plural))
|
|
112
|
+
i = end + len(close_m)
|
|
113
|
+
prose_start = i
|
|
114
|
+
if prose_start < n:
|
|
115
|
+
out.append(Fragment("prose", source[prose_start:], prose_start))
|
|
116
|
+
# A stray closing sigil in prose
|
|
117
|
+
for f in out:
|
|
118
|
+
if f.kind == "prose" and ">" in f.text:
|
|
119
|
+
raise UnmatchedMarker("unmatched '>'", position=f.pos + f.text.index(">"))
|
|
120
|
+
return out
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
_WORD_RE = re.compile(r"""
|
|
124
|
+
(?P<date>\d{4}-\d{2}-\d{2})(?![\w-])
|
|
125
|
+
| (?P<number>\d+(?:\.\d+)?)(?![\w-])(?!\.\d)
|
|
126
|
+
| (?P<poss>'s)(?![\w-])
|
|
127
|
+
| (?P<word>[^\s"()',.]+?)(?=[\s"()',.]|$|'s\b)
|
|
128
|
+
| (?P<punct>[(),.])
|
|
129
|
+
| (?P<space>\s+)
|
|
130
|
+
""", re.VERBOSE)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _prose_tokens(text: str, base: int) -> list[Token]:
|
|
134
|
+
out: list[Token] = []
|
|
135
|
+
i = 0
|
|
136
|
+
while i < len(text):
|
|
137
|
+
c = text[i]
|
|
138
|
+
if c == '"':
|
|
139
|
+
end = text.find('"', i + 1)
|
|
140
|
+
if end < 0:
|
|
141
|
+
raise UnterminatedString("unterminated string", position=base + i)
|
|
142
|
+
out.append(Token("string", text[i + 1:end], base + i))
|
|
143
|
+
i = end + 1
|
|
144
|
+
continue
|
|
145
|
+
m = _WORD_RE.match(text, i)
|
|
146
|
+
if not m:
|
|
147
|
+
raise UnmatchedMarker("unreadable text", position=base + i)
|
|
148
|
+
kind = m.lastgroup
|
|
149
|
+
if kind == "space":
|
|
150
|
+
pass
|
|
151
|
+
elif kind == "date":
|
|
152
|
+
out.append(Token("date", m.group(), base + i))
|
|
153
|
+
elif kind == "number":
|
|
154
|
+
s = m.group()
|
|
155
|
+
out.append(Token("number", float(s) if "." in s else int(s), base + i, raw=s))
|
|
156
|
+
elif kind == "poss":
|
|
157
|
+
out.append(Token("punct", "'s", base + i))
|
|
158
|
+
elif kind == "punct":
|
|
159
|
+
out.append(Token("punct", m.group(), base + i))
|
|
160
|
+
else:
|
|
161
|
+
out.append(Token("word", m.group().lower(), base + i, raw=m.group()))
|
|
162
|
+
i = m.end()
|
|
163
|
+
return out
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def tokens(source: str) -> tuple[list[Fragment], list[Token]]:
|
|
167
|
+
frags = fragments(source)
|
|
168
|
+
out: list[Token] = []
|
|
169
|
+
for f in frags:
|
|
170
|
+
if f.kind == "prose":
|
|
171
|
+
out.extend(_prose_tokens(f.text, f.pos))
|
|
172
|
+
elif f.kind == "quantity":
|
|
173
|
+
try:
|
|
174
|
+
out.append(Token("quantity", quantity(f.text), f.pos, raw=f.text))
|
|
175
|
+
except ValueError:
|
|
176
|
+
out.append(Token("quantity", None, f.pos, raw=f.text))
|
|
177
|
+
else:
|
|
178
|
+
out.append(Token(f.kind, f.text, f.pos, plural=f.plural))
|
|
179
|
+
depth = 0
|
|
180
|
+
first_open = None
|
|
181
|
+
for t in out:
|
|
182
|
+
if t.kind == "punct" and t.value == "(":
|
|
183
|
+
depth += 1
|
|
184
|
+
if first_open is None:
|
|
185
|
+
first_open = t.pos
|
|
186
|
+
elif t.kind == "punct" and t.value == ")":
|
|
187
|
+
depth -= 1
|
|
188
|
+
if depth < 0:
|
|
189
|
+
raise UnbalancedParenthesis("unbalanced parenthesis", position=t.pos)
|
|
190
|
+
if depth > 0:
|
|
191
|
+
raise UnbalancedParenthesis("unbalanced parenthesis", position=first_open)
|
|
192
|
+
if not out or not (out[-1].kind == "punct" and out[-1].value == "."):
|
|
193
|
+
raise MissingPeriod("a sentence ends with a period")
|
|
194
|
+
return frags, out
|