docketry 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docketry/__init__.py +8 -0
- docketry/cite.py +253 -0
- docketry/cite_client.py +68 -0
- docketry/classify.py +60 -0
- docketry/cli.py +634 -0
- docketry/config.py +77 -0
- docketry/envelope.py +166 -0
- docketry/extract.py +195 -0
- docketry/gates/__init__.py +28 -0
- docketry/gates/builtin.py +164 -0
- docketry/gates/classifier.py +28 -0
- docketry/gates/notice.py +59 -0
- docketry/lint.py +172 -0
- docketry/mailbox.py +77 -0
- docketry/manifest.py +105 -0
- docketry/notices.py +268 -0
- docketry/pipeline.py +157 -0
- docketry/store.py +340 -0
- docketry/webui.py +202 -0
- docketry-0.1.0.dist-info/METADATA +343 -0
- docketry-0.1.0.dist-info/RECORD +25 -0
- docketry-0.1.0.dist-info/WHEEL +4 -0
- docketry-0.1.0.dist-info/entry_points.txt +2 -0
- docketry-0.1.0.dist-info/licenses/LICENSE +202 -0
- docketry-0.1.0.dist-info/licenses/NOTICE +4 -0
docketry/gates/notice.py
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""Notice-parser gate: fail-loud recognition of court notification emails.
|
|
2
|
+
|
|
3
|
+
Not a match at all is fine (ordinary mail just flows on). A match that cannot
|
|
4
|
+
extract a field its adapter declared required is the loud failure: it means
|
|
5
|
+
the source changed its template, and the message parks for a human instead of
|
|
6
|
+
flowing through with silent holes.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from ..envelope import Envelope
|
|
14
|
+
from ..pipeline import Finding, SEVERITY_FAIL, SEVERITY_INFO
|
|
15
|
+
from .. import notices
|
|
16
|
+
from . import register
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@register
|
|
20
|
+
class NoticeParser:
|
|
21
|
+
id = "notice-parser"
|
|
22
|
+
allowed_stages = {"ingest"}
|
|
23
|
+
|
|
24
|
+
_cache: tuple[str | None, list] | None = None
|
|
25
|
+
|
|
26
|
+
def _adapters(self, options: dict) -> list:
|
|
27
|
+
adapters_file = options.get("adapters_file")
|
|
28
|
+
if adapters_file and not Path(adapters_file).exists():
|
|
29
|
+
raise notices.AdapterError(f"adapters_file not found: {adapters_file}")
|
|
30
|
+
cache = type(self)._cache
|
|
31
|
+
if cache is not None and cache[0] == adapters_file:
|
|
32
|
+
return cache[1]
|
|
33
|
+
stack = notices.stack(adapters_file)
|
|
34
|
+
type(self)._cache = (adapters_file, stack)
|
|
35
|
+
return stack
|
|
36
|
+
|
|
37
|
+
def check(self, env: Envelope, options: dict) -> list[Finding]:
|
|
38
|
+
result = notices.parse(env, self._adapters(options))
|
|
39
|
+
if result is None:
|
|
40
|
+
return []
|
|
41
|
+
findings = [
|
|
42
|
+
Finding(
|
|
43
|
+
self.id,
|
|
44
|
+
SEVERITY_INFO,
|
|
45
|
+
f"{result.notice_type} via {result.adapter}: "
|
|
46
|
+
+ json.dumps(result.fields, ensure_ascii=False)[:400],
|
|
47
|
+
)
|
|
48
|
+
]
|
|
49
|
+
for fname in result.missing:
|
|
50
|
+
findings.append(
|
|
51
|
+
Finding(
|
|
52
|
+
self.id,
|
|
53
|
+
SEVERITY_FAIL,
|
|
54
|
+
f"adapter '{result.adapter}' matched but required field"
|
|
55
|
+
f" '{fname}' did not extract — the source may have changed"
|
|
56
|
+
" its template",
|
|
57
|
+
)
|
|
58
|
+
)
|
|
59
|
+
return findings
|
docketry/lint.py
ADDED
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
"""Brief linter: deterministic writing checks for litigation drafts.
|
|
2
|
+
|
|
3
|
+
Lint drafts like code: cheap, deterministic checks run before a human spends
|
|
4
|
+
review time — the human reviews findings, not raw text. Every rule here is
|
|
5
|
+
an editor, not a lawyer: nothing asserts what the law is, only what the
|
|
6
|
+
document does to itself.
|
|
7
|
+
|
|
8
|
+
Built-in rules:
|
|
9
|
+
- credibility-language: in a summary-judgment context, words that accuse the
|
|
10
|
+
other side of lying (misrepresented, concealed, intentionally, conveniently
|
|
11
|
+
omitted) invite "that's a jury question" — flagged only when the draft is
|
|
12
|
+
SJ briefing.
|
|
13
|
+
- uncited-testimony: a line asserting sworn testimony with no record pin cite
|
|
14
|
+
on that line.
|
|
15
|
+
- date-contradiction: the draft asserts a deadline closed/expired on a date
|
|
16
|
+
that is AFTER the draft's own certificate-of-service date.
|
|
17
|
+
- reporter-spacing: So.2d/So.3d written without the space (So. 2d / So. 3d).
|
|
18
|
+
|
|
19
|
+
Firm rulepacks are TOML: pattern rules with an id, message, and severity,
|
|
20
|
+
validated at load. Packs are shareable the way linter configs are.
|
|
21
|
+
"""
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import re
|
|
25
|
+
import tomllib
|
|
26
|
+
from dataclasses import dataclass
|
|
27
|
+
from datetime import datetime
|
|
28
|
+
from pathlib import Path
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class RulepackError(ValueError):
|
|
32
|
+
pass
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass
|
|
36
|
+
class LintFinding:
|
|
37
|
+
rule: str
|
|
38
|
+
severity: str # error | warn
|
|
39
|
+
line: int # 1-based; 0 = document-level
|
|
40
|
+
message: str
|
|
41
|
+
excerpt: str = ""
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
_SJ_CONTEXT = re.compile(
|
|
45
|
+
r"summary\s+judgment|rule\s+1\.510|fed\.?\s*r\.?\s*civ\.?\s*p\.?\s*56|rule\s+56\b",
|
|
46
|
+
re.IGNORECASE,
|
|
47
|
+
)
|
|
48
|
+
_CREDIBILITY = re.compile(
|
|
49
|
+
r"\b(misrepresent(?:s|ed|ation|ing)?|conceal(?:s|ed|ment|ing)?|"
|
|
50
|
+
r"intentionally\s+(?:misled|omitted|hid)|conveniently\s+omit(?:s|ted)?|"
|
|
51
|
+
r"lie[sd]?\b|lying|self-serving)\b",
|
|
52
|
+
re.IGNORECASE,
|
|
53
|
+
)
|
|
54
|
+
_TESTIMONY = re.compile(
|
|
55
|
+
r"\b(testified|testifies|swore|sworn\s+testimony|admitted\s+(?:at|in|during)|"
|
|
56
|
+
r"stated\s+(?:at|in)\s+(?:his|her|their)\s+deposition)\b",
|
|
57
|
+
re.IGNORECASE,
|
|
58
|
+
)
|
|
59
|
+
_PIN_CITE = re.compile(
|
|
60
|
+
r"(\bEx(?:h)?\.\s|\bExhibit\s+[A-Z0-9]|¶|\bp\.\s*\d|\bpp\.\s*\d|"
|
|
61
|
+
r"\bat\s+\d+|\d+:\d+|\bDep\.|\bT\.\s*(?:at\s*)?\d|\bR\.\s*(?:at\s*)?\d)",
|
|
62
|
+
)
|
|
63
|
+
_REPORTER_SPACING = re.compile(r"\bSo\.([23])d\b")
|
|
64
|
+
|
|
65
|
+
_DATE_RX = re.compile(
|
|
66
|
+
r"((?:January|February|March|April|May|June|July|August|September|October|"
|
|
67
|
+
r"November|December)\s+\d{1,2},\s+\d{4}|\d{1,2}/\d{1,2}/\d{4})"
|
|
68
|
+
)
|
|
69
|
+
_CLOSED_RX = re.compile(
|
|
70
|
+
r"(?:discovery|disclosure)s?\s+(?:period\s+)?(?:has\s+)?"
|
|
71
|
+
r"(?:closed|expired|ended|passed)\s+on\s+" + _DATE_RX.pattern,
|
|
72
|
+
re.IGNORECASE,
|
|
73
|
+
)
|
|
74
|
+
_CERT_RX = re.compile(r"certificate\s+of\s+service", re.IGNORECASE)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _parse_date(s: str) -> datetime | None:
|
|
78
|
+
for fmt in ("%B %d, %Y", "%m/%d/%Y"):
|
|
79
|
+
try:
|
|
80
|
+
return datetime.strptime(s, fmt)
|
|
81
|
+
except ValueError:
|
|
82
|
+
continue
|
|
83
|
+
return None
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _is_heading(line: str) -> bool:
|
|
87
|
+
stripped = line.strip()
|
|
88
|
+
return bool(stripped) and stripped == stripped.upper() and len(stripped) < 80
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def lint(text: str, rulepack: list[dict] | None = None) -> list[LintFinding]:
|
|
92
|
+
lines = text.splitlines()
|
|
93
|
+
findings: list[LintFinding] = []
|
|
94
|
+
sj = bool(_SJ_CONTEXT.search(text))
|
|
95
|
+
|
|
96
|
+
for n, line in enumerate(lines, start=1):
|
|
97
|
+
if _is_heading(line):
|
|
98
|
+
continue
|
|
99
|
+
if sj:
|
|
100
|
+
m = _CREDIBILITY.search(line)
|
|
101
|
+
if m and '"' not in line[: m.start()]: # quoted testimony is fair game
|
|
102
|
+
findings.append(LintFinding(
|
|
103
|
+
"credibility-language", "warn", n,
|
|
104
|
+
f"'{m.group(0)}' in summary-judgment briefing invites a"
|
|
105
|
+
" credibility question a court cannot resolve on SJ —"
|
|
106
|
+
" recast as absence of competent evidence",
|
|
107
|
+
line.strip()[:120],
|
|
108
|
+
))
|
|
109
|
+
if _TESTIMONY.search(line) and not _PIN_CITE.search(line):
|
|
110
|
+
findings.append(LintFinding(
|
|
111
|
+
"uncited-testimony", "error", n,
|
|
112
|
+
"sworn-testimony assertion with no record pin cite on the line",
|
|
113
|
+
line.strip()[:120],
|
|
114
|
+
))
|
|
115
|
+
for m in _REPORTER_SPACING.finditer(line):
|
|
116
|
+
findings.append(LintFinding(
|
|
117
|
+
"reporter-spacing", "warn", n,
|
|
118
|
+
f"'So.{m.group(1)}d' should be 'So. {m.group(1)}d'",
|
|
119
|
+
line.strip()[:120],
|
|
120
|
+
))
|
|
121
|
+
|
|
122
|
+
closed = _CLOSED_RX.search(text)
|
|
123
|
+
if closed:
|
|
124
|
+
closed_date = _parse_date(closed.group(1))
|
|
125
|
+
cert = _CERT_RX.search(text)
|
|
126
|
+
if closed_date and cert:
|
|
127
|
+
cert_dates = _DATE_RX.findall(text[cert.start():cert.start() + 600])
|
|
128
|
+
for ds in cert_dates:
|
|
129
|
+
cert_date = _parse_date(ds)
|
|
130
|
+
if cert_date and cert_date < closed_date:
|
|
131
|
+
findings.append(LintFinding(
|
|
132
|
+
"date-contradiction", "error", 0,
|
|
133
|
+
f"the draft says discovery closed on"
|
|
134
|
+
f" {closed.group(1)} but its certificate of service"
|
|
135
|
+
f" is dated {ds} — it asserts a past event that has"
|
|
136
|
+
" not happened yet",
|
|
137
|
+
))
|
|
138
|
+
break
|
|
139
|
+
|
|
140
|
+
for rule in rulepack or []:
|
|
141
|
+
rx = rule["_compiled"]
|
|
142
|
+
for n, line in enumerate(lines, start=1):
|
|
143
|
+
m = rx.search(line)
|
|
144
|
+
if m:
|
|
145
|
+
findings.append(LintFinding(
|
|
146
|
+
rule["id"], rule.get("severity", "warn"), n,
|
|
147
|
+
rule["message"], line.strip()[:120],
|
|
148
|
+
))
|
|
149
|
+
findings.sort(key=lambda f: (f.line or 10**9, f.rule))
|
|
150
|
+
return findings
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def load_rulepack(path: str | Path) -> list[dict]:
|
|
154
|
+
data = tomllib.loads(Path(path).read_text())
|
|
155
|
+
rules = []
|
|
156
|
+
for i, r in enumerate(data.get("rule", [])):
|
|
157
|
+
rid = r.get("id")
|
|
158
|
+
if not rid:
|
|
159
|
+
raise RulepackError(f"rule #{i + 1} has no id")
|
|
160
|
+
if r.get("severity", "warn") not in ("error", "warn"):
|
|
161
|
+
raise RulepackError(f"rule '{rid}': severity must be error or warn")
|
|
162
|
+
pattern = r.get("pattern")
|
|
163
|
+
if not pattern:
|
|
164
|
+
raise RulepackError(f"rule '{rid}' has no pattern")
|
|
165
|
+
if not r.get("message"):
|
|
166
|
+
raise RulepackError(f"rule '{rid}' has no message")
|
|
167
|
+
try:
|
|
168
|
+
r["_compiled"] = re.compile(pattern, re.IGNORECASE)
|
|
169
|
+
except re.error as e:
|
|
170
|
+
raise RulepackError(f"rule '{rid}': pattern does not compile: {e}")
|
|
171
|
+
rules.append(r)
|
|
172
|
+
return rules
|
docketry/mailbox.py
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""Drain a dedicated intake mailbox over IMAP.
|
|
2
|
+
|
|
3
|
+
Plan-1 intake: the firm creates a mailbox that exists solely to be the port,
|
|
4
|
+
points forwarding rules at it, and this module reads it with credentials the
|
|
5
|
+
firm holds. Strictly read-only: the folder is opened readonly, nothing is
|
|
6
|
+
marked, moved, or deleted, and a UID cursor (with UIDVALIDITY tracking) makes
|
|
7
|
+
every sweep idempotent.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import imaplib
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
from typing import Iterator
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass
|
|
17
|
+
class MailboxConfig:
|
|
18
|
+
host: str
|
|
19
|
+
user: str
|
|
20
|
+
password: str
|
|
21
|
+
folder: str = "INBOX"
|
|
22
|
+
port: int = 993
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class IntakeMailbox:
|
|
26
|
+
def __init__(self, cfg: MailboxConfig):
|
|
27
|
+
self.cfg = cfg
|
|
28
|
+
self.conn: imaplib.IMAP4_SSL | None = None
|
|
29
|
+
|
|
30
|
+
def __enter__(self) -> "IntakeMailbox":
|
|
31
|
+
self.conn = imaplib.IMAP4_SSL(self.cfg.host, self.cfg.port)
|
|
32
|
+
self.conn.login(self.cfg.user, self.cfg.password)
|
|
33
|
+
typ, _ = self.conn.select(self.cfg.folder, readonly=True)
|
|
34
|
+
if typ != "OK":
|
|
35
|
+
raise RuntimeError(f"cannot open folder {self.cfg.folder!r}")
|
|
36
|
+
return self
|
|
37
|
+
|
|
38
|
+
def __exit__(self, *exc) -> None:
|
|
39
|
+
if self.conn is not None:
|
|
40
|
+
try:
|
|
41
|
+
self.conn.logout()
|
|
42
|
+
except Exception:
|
|
43
|
+
pass
|
|
44
|
+
self.conn = None
|
|
45
|
+
|
|
46
|
+
@property
|
|
47
|
+
def label(self) -> str:
|
|
48
|
+
return f"{self.cfg.user}/{self.cfg.folder}@{self.cfg.host}"
|
|
49
|
+
|
|
50
|
+
def uidvalidity(self) -> int | None:
|
|
51
|
+
typ, data = self.conn.response("UIDVALIDITY")
|
|
52
|
+
try:
|
|
53
|
+
return int(data[0])
|
|
54
|
+
except (TypeError, ValueError, IndexError):
|
|
55
|
+
return None
|
|
56
|
+
|
|
57
|
+
def new_messages(self, last_uid: int) -> Iterator[tuple[int, bytes]]:
|
|
58
|
+
"""Yield (uid, raw_rfc822) for every message with uid > last_uid."""
|
|
59
|
+
typ, data = self.conn.uid("SEARCH", None, f"UID {last_uid + 1}:*")
|
|
60
|
+
if typ != "OK" or not data or not data[0]:
|
|
61
|
+
return
|
|
62
|
+
for uid_b in data[0].split():
|
|
63
|
+
uid = int(uid_b)
|
|
64
|
+
if uid <= last_uid:
|
|
65
|
+
# IMAP quirk: "n:*" matches the highest-numbered message even
|
|
66
|
+
# when n exceeds it; skip anything we've already swept.
|
|
67
|
+
continue
|
|
68
|
+
typ, msg_data = self.conn.uid("FETCH", str(uid), "(RFC822)")
|
|
69
|
+
if typ != "OK" or not msg_data or msg_data[0] is None:
|
|
70
|
+
continue
|
|
71
|
+
raw = None
|
|
72
|
+
for part in msg_data:
|
|
73
|
+
if isinstance(part, tuple) and len(part) >= 2:
|
|
74
|
+
raw = part[1]
|
|
75
|
+
break
|
|
76
|
+
if raw:
|
|
77
|
+
yield uid, raw
|
docketry/manifest.py
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""Load and validate the guardrail manifest (TOML).
|
|
2
|
+
|
|
3
|
+
The manifest is where a firm declares its pipeline stages and which gate runs
|
|
4
|
+
where. Validation is strict and load-time: unknown gates, unknown stages, a
|
|
5
|
+
gate bound outside its declared allowed_stages, or a bare on_fail value all
|
|
6
|
+
refuse to load — a misconfigured pipeline never runs half-enforced.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import tomllib
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from . import gates as gate_registry
|
|
14
|
+
from .pipeline import GateBinding, ON_FAIL, Pipeline
|
|
15
|
+
|
|
16
|
+
DEFAULT_MANIFEST = """\
|
|
17
|
+
# Docketry guardrail manifest.
|
|
18
|
+
# Stages run left to right; each [[gate]] declares where it binds, what
|
|
19
|
+
# happens on failure (block | bounce | warn), and which role can approve.
|
|
20
|
+
|
|
21
|
+
[pipeline]
|
|
22
|
+
stages = ["ingest", "review"]
|
|
23
|
+
|
|
24
|
+
[[gate]]
|
|
25
|
+
id = "attachment-policy"
|
|
26
|
+
binds_to = ["ingest"]
|
|
27
|
+
on_fail = "bounce"
|
|
28
|
+
authority = "paralegal"
|
|
29
|
+
|
|
30
|
+
[gate.options]
|
|
31
|
+
max_size_mb = 25
|
|
32
|
+
|
|
33
|
+
[[gate]]
|
|
34
|
+
id = "provenance-stamp"
|
|
35
|
+
binds_to = ["ingest"]
|
|
36
|
+
on_fail = "warn"
|
|
37
|
+
authority = "paralegal"
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class ManifestError(ValueError):
|
|
42
|
+
pass
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def load_manifest(path: str | Path) -> Pipeline:
|
|
46
|
+
data = tomllib.loads(Path(path).read_text())
|
|
47
|
+
return build_pipeline(data)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def build_pipeline(data: dict) -> Pipeline:
|
|
51
|
+
stages = data.get("pipeline", {}).get("stages", [])
|
|
52
|
+
if not stages or not all(isinstance(s, str) and s for s in stages):
|
|
53
|
+
raise ManifestError("[pipeline].stages must be a non-empty list of stage names")
|
|
54
|
+
if len(set(stages)) != len(stages):
|
|
55
|
+
raise ManifestError("duplicate stage names in [pipeline].stages")
|
|
56
|
+
|
|
57
|
+
bindings: list[GateBinding] = []
|
|
58
|
+
for i, g in enumerate(data.get("gate", [])):
|
|
59
|
+
gid = g.get("id")
|
|
60
|
+
if not gid:
|
|
61
|
+
raise ManifestError(f"gate #{i + 1} has no id")
|
|
62
|
+
try:
|
|
63
|
+
cls = gate_registry.get(gid)
|
|
64
|
+
except KeyError as e:
|
|
65
|
+
raise ManifestError(str(e)) from None
|
|
66
|
+
|
|
67
|
+
binds_to = g.get("binds_to", [])
|
|
68
|
+
if not binds_to:
|
|
69
|
+
raise ManifestError(f"gate '{gid}' declares no binds_to stages")
|
|
70
|
+
unknown = [s for s in binds_to if s not in stages]
|
|
71
|
+
if unknown:
|
|
72
|
+
raise ManifestError(f"gate '{gid}' binds to unknown stage(s): {unknown}")
|
|
73
|
+
|
|
74
|
+
allowed = getattr(cls, "allowed_stages", None)
|
|
75
|
+
if allowed is not None:
|
|
76
|
+
out_of_scope = [s for s in binds_to if s not in allowed]
|
|
77
|
+
if out_of_scope:
|
|
78
|
+
raise ManifestError(
|
|
79
|
+
f"gate '{gid}' is not meant for stage(s) {out_of_scope};"
|
|
80
|
+
f" it belongs in: {sorted(allowed)}"
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
on_fail = g.get("on_fail", "bounce")
|
|
84
|
+
if on_fail not in ON_FAIL:
|
|
85
|
+
raise ManifestError(
|
|
86
|
+
f"gate '{gid}' has on_fail='{on_fail}' (must be one of {ON_FAIL})"
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
gate = cls()
|
|
90
|
+
validator = getattr(gate, "validate_options", None)
|
|
91
|
+
if validator is not None:
|
|
92
|
+
problems = validator(g.get("options", {}))
|
|
93
|
+
if problems:
|
|
94
|
+
raise ManifestError(f"gate '{gid}' options invalid: {'; '.join(problems)}")
|
|
95
|
+
|
|
96
|
+
bindings.append(
|
|
97
|
+
GateBinding(
|
|
98
|
+
gate=gate,
|
|
99
|
+
binds_to=list(binds_to),
|
|
100
|
+
on_fail=on_fail,
|
|
101
|
+
authority=g.get("authority", "attorney"),
|
|
102
|
+
options=g.get("options", {}),
|
|
103
|
+
)
|
|
104
|
+
)
|
|
105
|
+
return Pipeline(stages=list(stages), bindings=bindings)
|
docketry/notices.py
ADDED
|
@@ -0,0 +1,268 @@
|
|
|
1
|
+
"""Court-notice parsing: adapters turn notification emails into typed notices.
|
|
2
|
+
|
|
3
|
+
An adapter recognizes one source system's notification email (by sender and
|
|
4
|
+
format fingerprint) and extracts its facts into a common schema. Built-in
|
|
5
|
+
adapters cover the big systems; firms add their own local courts as TOML
|
|
6
|
+
config — no code — and firm adapters are consulted first so a local template
|
|
7
|
+
can override a built-in.
|
|
8
|
+
|
|
9
|
+
Design rules, enforced here:
|
|
10
|
+
- Format parsing only. An adapter that matches but cannot extract a field it
|
|
11
|
+
declared required reports the miss loudly (the gate bounces the message)
|
|
12
|
+
instead of passing a notice through with silent holes.
|
|
13
|
+
- Capture, never consume. A PACER NEF's one-time "free look" document link is
|
|
14
|
+
extracted as data; nothing in this module fetches URLs, ever — an automated
|
|
15
|
+
fetch would silently burn the single free access.
|
|
16
|
+
- Extraction is open; judgment is not. A hearing notice yields date, time,
|
|
17
|
+
judge, location. What lands on whose calendar is a human decision that
|
|
18
|
+
happens downstream of this code.
|
|
19
|
+
"""
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import re
|
|
23
|
+
import tomllib
|
|
24
|
+
from dataclasses import dataclass, field
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
|
|
27
|
+
from .envelope import Envelope
|
|
28
|
+
|
|
29
|
+
NOTICE_TYPES = ("service_notice", "filing_receipt", "hearing_notice")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class AdapterError(ValueError):
|
|
33
|
+
"""A misdeclared adapter refuses to load; it never half-parses."""
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass
|
|
37
|
+
class NoticeResult:
|
|
38
|
+
adapter: str
|
|
39
|
+
notice_type: str
|
|
40
|
+
fields: dict
|
|
41
|
+
missing: list[str] = field(default_factory=list)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _searchable(env: Envelope) -> str:
|
|
45
|
+
return f"{env.subject}\n{env.body_text}"
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@dataclass
|
|
49
|
+
class PatternAdapter:
|
|
50
|
+
"""One source system: match rules + anchored field patterns.
|
|
51
|
+
|
|
52
|
+
Both built-in and TOML-defined adapters are instances of this class, so
|
|
53
|
+
a firm's config adapter has exactly the powers of a shipped one.
|
|
54
|
+
"""
|
|
55
|
+
|
|
56
|
+
name: str
|
|
57
|
+
notice_type: str
|
|
58
|
+
from_endswith: tuple[str, ...] = ()
|
|
59
|
+
subject_contains: tuple[str, ...] = ()
|
|
60
|
+
body_contains: tuple[str, ...] = ()
|
|
61
|
+
fields: dict[str, re.Pattern] = field(default_factory=dict)
|
|
62
|
+
list_fields: dict[str, re.Pattern] = field(default_factory=dict)
|
|
63
|
+
required: tuple[str, ...] = ()
|
|
64
|
+
|
|
65
|
+
def __post_init__(self) -> None:
|
|
66
|
+
if self.notice_type not in NOTICE_TYPES:
|
|
67
|
+
raise AdapterError(
|
|
68
|
+
f"adapter '{self.name}': notice_type '{self.notice_type}'"
|
|
69
|
+
f" is not one of {NOTICE_TYPES}"
|
|
70
|
+
)
|
|
71
|
+
if not (self.from_endswith or self.subject_contains or self.body_contains):
|
|
72
|
+
raise AdapterError(f"adapter '{self.name}' declares no match rules")
|
|
73
|
+
for fname in self.required:
|
|
74
|
+
if fname not in self.fields and fname not in self.list_fields:
|
|
75
|
+
raise AdapterError(
|
|
76
|
+
f"adapter '{self.name}' requires field '{fname}' but has no pattern for it"
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
def match(self, env: Envelope) -> bool:
|
|
80
|
+
sender = env.from_addr.lower()
|
|
81
|
+
if self.from_endswith and not any(sender.endswith(s.lower()) for s in self.from_endswith):
|
|
82
|
+
return False
|
|
83
|
+
subject = env.subject.lower()
|
|
84
|
+
if self.subject_contains and not any(s.lower() in subject for s in self.subject_contains):
|
|
85
|
+
return False
|
|
86
|
+
if self.body_contains:
|
|
87
|
+
body = env.body_text.lower()
|
|
88
|
+
if not any(s.lower() in body for s in self.body_contains):
|
|
89
|
+
return False
|
|
90
|
+
return True
|
|
91
|
+
|
|
92
|
+
def extract(self, env: Envelope) -> NoticeResult:
|
|
93
|
+
text = _searchable(env)
|
|
94
|
+
out: dict = {}
|
|
95
|
+
for fname, pat in self.fields.items():
|
|
96
|
+
m = pat.search(text)
|
|
97
|
+
if m:
|
|
98
|
+
out[fname] = m.group(1).strip()
|
|
99
|
+
for fname, pat in self.list_fields.items():
|
|
100
|
+
hits = [h.strip() for h in pat.findall(text) if h.strip()]
|
|
101
|
+
if hits:
|
|
102
|
+
out[fname] = hits
|
|
103
|
+
missing = [f for f in self.required if f not in out]
|
|
104
|
+
return NoticeResult(
|
|
105
|
+
adapter=self.name, notice_type=self.notice_type, fields=out, missing=missing
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _rx(pattern: str, *, name: str, fname: str) -> re.Pattern:
|
|
110
|
+
try:
|
|
111
|
+
pat = re.compile(pattern, re.IGNORECASE | re.MULTILINE)
|
|
112
|
+
except re.error as e:
|
|
113
|
+
raise AdapterError(f"adapter '{name}': field '{fname}' pattern does not compile: {e}")
|
|
114
|
+
if pat.groups != 1:
|
|
115
|
+
raise AdapterError(
|
|
116
|
+
f"adapter '{name}': field '{fname}' pattern must have exactly one capture group"
|
|
117
|
+
)
|
|
118
|
+
return pat
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
# ---------------------------------------------------------------------------
|
|
122
|
+
# Built-in adapters. Template fingerprints for the major systems; formats
|
|
123
|
+
# vary by district/circuit, so patterns are anchored and conservative, and a
|
|
124
|
+
# firm TOML adapter can always override (firm adapters run first).
|
|
125
|
+
# ---------------------------------------------------------------------------
|
|
126
|
+
|
|
127
|
+
def builtin_adapters() -> list[PatternAdapter]:
|
|
128
|
+
return [
|
|
129
|
+
PatternAdapter(
|
|
130
|
+
name="fl-eportal-service",
|
|
131
|
+
notice_type="service_notice",
|
|
132
|
+
from_endswith=("@myflcourtaccess.com",),
|
|
133
|
+
subject_contains=("service of court document", "notice of service"),
|
|
134
|
+
fields={
|
|
135
|
+
"case_number": _rx(r"^\s*Case\s*(?:Number|#)\s*:?\s*([A-Z0-9-]{6,25})",
|
|
136
|
+
name="fl-eportal-service", fname="case_number"),
|
|
137
|
+
"case_style": _rx(r"^\s*Case Style\s*:?\s*(.+)",
|
|
138
|
+
name="fl-eportal-service", fname="case_style"),
|
|
139
|
+
"court": _rx(r"^\s*Court\s*:?\s*(.+)", name="fl-eportal-service", fname="court"),
|
|
140
|
+
},
|
|
141
|
+
list_fields={
|
|
142
|
+
"documents": _rx(r"^\s*Document(?:s)?\s*:?\s*(.+)$",
|
|
143
|
+
name="fl-eportal-service", fname="documents"),
|
|
144
|
+
"served": _rx(r"^\s*Served\s*:?\s*(.+@.+)$",
|
|
145
|
+
name="fl-eportal-service", fname="served"),
|
|
146
|
+
},
|
|
147
|
+
required=("case_number",),
|
|
148
|
+
),
|
|
149
|
+
PatternAdapter(
|
|
150
|
+
name="pacer-nef",
|
|
151
|
+
notice_type="service_notice",
|
|
152
|
+
from_endswith=("uscourts.gov",),
|
|
153
|
+
subject_contains=("activity in case",),
|
|
154
|
+
fields={
|
|
155
|
+
"case_number": _rx(r"Activity in Case\s+(\S+)",
|
|
156
|
+
name="pacer-nef", fname="case_number"),
|
|
157
|
+
"case_name": _rx(r"Activity in Case\s+\S+\s+(.+?)(?:\s{2,}|$)",
|
|
158
|
+
name="pacer-nef", fname="case_name"),
|
|
159
|
+
"docket_text": _rx(r"^\s*Docket Text\s*:?\s*(.+)",
|
|
160
|
+
name="pacer-nef", fname="docket_text"),
|
|
161
|
+
"document_number": _rx(r"^\s*Document Number\s*:?\s*(\d+)",
|
|
162
|
+
name="pacer-nef", fname="document_number"),
|
|
163
|
+
# Captured as data only. NEVER fetched: the NEF link is the
|
|
164
|
+
# recipient's one-time free look, and an automated fetch
|
|
165
|
+
# silently consumes it.
|
|
166
|
+
"document_link": _rx(r"(https?://ecf\.\S*uscourts\.gov/\S+)",
|
|
167
|
+
name="pacer-nef", fname="document_link"),
|
|
168
|
+
},
|
|
169
|
+
required=("case_number",),
|
|
170
|
+
),
|
|
171
|
+
PatternAdapter(
|
|
172
|
+
name="jacs-hearing",
|
|
173
|
+
notice_type="hearing_notice",
|
|
174
|
+
body_contains=("judicial automated calendaring system", "jacs"),
|
|
175
|
+
fields={
|
|
176
|
+
"hearing_date": _rx(r"^\s*(?:Hearing\s+)?Date\s*:?\s*(\d{1,2}/\d{1,2}/\d{2,4})",
|
|
177
|
+
name="jacs-hearing", fname="hearing_date"),
|
|
178
|
+
"hearing_time": _rx(r"^\s*Time\s*:?\s*(\d{1,2}:\d{2}\s*(?:AM|PM)?)",
|
|
179
|
+
name="jacs-hearing", fname="hearing_time"),
|
|
180
|
+
"judge": _rx(r"^\s*Judge\s*:?\s*(.+)", name="jacs-hearing", fname="judge"),
|
|
181
|
+
"case_number": _rx(r"^\s*Case\s*(?:Number|#|No\.?)\s*:?\s*([A-Z0-9-]{6,25})",
|
|
182
|
+
name="jacs-hearing", fname="case_number"),
|
|
183
|
+
"matter": _rx(r"^\s*Matter\s*:?\s*(.+)", name="jacs-hearing", fname="matter"),
|
|
184
|
+
},
|
|
185
|
+
required=("hearing_date",),
|
|
186
|
+
),
|
|
187
|
+
PatternAdapter(
|
|
188
|
+
name="jaws-hearing",
|
|
189
|
+
notice_type="hearing_notice",
|
|
190
|
+
body_contains=("judicial automated workflow system", "jaws"),
|
|
191
|
+
fields={
|
|
192
|
+
"hearing_date": _rx(r"^\s*(?:Hearing\s+)?Date\s*:?\s*(\d{1,2}/\d{1,2}/\d{2,4})",
|
|
193
|
+
name="jaws-hearing", fname="hearing_date"),
|
|
194
|
+
"hearing_time": _rx(r"^\s*Time\s*:?\s*(\d{1,2}:\d{2}\s*(?:AM|PM)?)",
|
|
195
|
+
name="jaws-hearing", fname="hearing_time"),
|
|
196
|
+
"judge": _rx(r"^\s*Judge\s*:?\s*(.+)", name="jaws-hearing", fname="judge"),
|
|
197
|
+
"case_number": _rx(r"^\s*Case\s*(?:Number|#|No\.?)\s*:?\s*([A-Z0-9-]{6,25})",
|
|
198
|
+
name="jaws-hearing", fname="case_number"),
|
|
199
|
+
},
|
|
200
|
+
required=("hearing_date",),
|
|
201
|
+
),
|
|
202
|
+
PatternAdapter(
|
|
203
|
+
name="efile-receipt",
|
|
204
|
+
notice_type="filing_receipt",
|
|
205
|
+
from_endswith=("efilingmail.tylertech.cloud", "tylerhost.net"),
|
|
206
|
+
subject_contains=("filing",),
|
|
207
|
+
fields={
|
|
208
|
+
"envelope_number": _rx(r"^\s*Envelope\s*(?:Number|#)\s*:?\s*(\d+)",
|
|
209
|
+
name="efile-receipt", fname="envelope_number"),
|
|
210
|
+
"case_number": _rx(r"^\s*Case\s*(?:Number|#)\s*:?\s*([A-Z0-9-]{6,25})",
|
|
211
|
+
name="efile-receipt", fname="case_number"),
|
|
212
|
+
"status": _rx(r"\bFiling\s+(Accepted|Submitted|Rejected|Returned)\b",
|
|
213
|
+
name="efile-receipt", fname="status"),
|
|
214
|
+
"filing_description": _rx(r"^\s*Filing\s+(?:Description|Type)\s*:?\s*(.+)",
|
|
215
|
+
name="efile-receipt", fname="filing_description"),
|
|
216
|
+
},
|
|
217
|
+
required=("envelope_number",),
|
|
218
|
+
),
|
|
219
|
+
]
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
# ---------------------------------------------------------------------------
|
|
223
|
+
# Firm-defined adapters: TOML in, PatternAdapter out, validated at load.
|
|
224
|
+
# ---------------------------------------------------------------------------
|
|
225
|
+
|
|
226
|
+
def load_adapters_file(path: str | Path) -> list[PatternAdapter]:
|
|
227
|
+
data = tomllib.loads(Path(path).read_text())
|
|
228
|
+
adapters: list[PatternAdapter] = []
|
|
229
|
+
for i, a in enumerate(data.get("adapter", [])):
|
|
230
|
+
name = a.get("name")
|
|
231
|
+
if not name:
|
|
232
|
+
raise AdapterError(f"adapter #{i + 1} has no name")
|
|
233
|
+
match = a.get("match", {})
|
|
234
|
+
|
|
235
|
+
def _tuple(val) -> tuple[str, ...]:
|
|
236
|
+
if val is None:
|
|
237
|
+
return ()
|
|
238
|
+
return (val,) if isinstance(val, str) else tuple(val)
|
|
239
|
+
|
|
240
|
+
adapters.append(
|
|
241
|
+
PatternAdapter(
|
|
242
|
+
name=name,
|
|
243
|
+
notice_type=a.get("notice_type", ""),
|
|
244
|
+
from_endswith=_tuple(match.get("from")),
|
|
245
|
+
subject_contains=_tuple(match.get("subject_contains")),
|
|
246
|
+
body_contains=_tuple(match.get("body_contains")),
|
|
247
|
+
fields={
|
|
248
|
+
fname: _rx(pat, name=name, fname=fname)
|
|
249
|
+
for fname, pat in a.get("fields", {}).items()
|
|
250
|
+
},
|
|
251
|
+
required=tuple(a.get("required", [])),
|
|
252
|
+
)
|
|
253
|
+
)
|
|
254
|
+
return adapters
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def parse(env: Envelope, adapters: list[PatternAdapter]) -> NoticeResult | None:
|
|
258
|
+
"""First matching adapter wins; callers put firm adapters before built-ins."""
|
|
259
|
+
for adapter in adapters:
|
|
260
|
+
if adapter.match(env):
|
|
261
|
+
return adapter.extract(env)
|
|
262
|
+
return None
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def stack(adapters_file: str | Path | None = None) -> list[PatternAdapter]:
|
|
266
|
+
"""Firm adapters (if any) first, then built-ins."""
|
|
267
|
+
firm = load_adapters_file(adapters_file) if adapters_file else []
|
|
268
|
+
return firm + builtin_adapters()
|