docketry 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docketry/__init__.py +8 -0
- docketry/cite.py +253 -0
- docketry/cite_client.py +68 -0
- docketry/classify.py +60 -0
- docketry/cli.py +634 -0
- docketry/config.py +77 -0
- docketry/envelope.py +166 -0
- docketry/extract.py +195 -0
- docketry/gates/__init__.py +28 -0
- docketry/gates/builtin.py +164 -0
- docketry/gates/classifier.py +28 -0
- docketry/gates/notice.py +59 -0
- docketry/lint.py +172 -0
- docketry/mailbox.py +77 -0
- docketry/manifest.py +105 -0
- docketry/notices.py +268 -0
- docketry/pipeline.py +157 -0
- docketry/store.py +340 -0
- docketry/webui.py +202 -0
- docketry-0.1.0.dist-info/METADATA +343 -0
- docketry-0.1.0.dist-info/RECORD +25 -0
- docketry-0.1.0.dist-info/WHEEL +4 -0
- docketry-0.1.0.dist-info/entry_points.txt +2 -0
- docketry-0.1.0.dist-info/licenses/LICENSE +202 -0
- docketry-0.1.0.dist-info/licenses/NOTICE +4 -0
docketry/config.py
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""Home-directory config for a Docketry installation.
|
|
2
|
+
|
|
3
|
+
A Docketry "home" is one directory holding config.toml, the guardrail
|
|
4
|
+
manifest, the SQLite store, and attachments — the whole installation is one
|
|
5
|
+
folder on the firm's own disk. The IMAP password is read from the
|
|
6
|
+
DOCKETRY_IMAP_PASSWORD environment variable first; storing it in config.toml
|
|
7
|
+
is supported for single-machine setups (the file is chmod 0600 on write).
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import os
|
|
12
|
+
import stat
|
|
13
|
+
import tomllib
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
from .mailbox import MailboxConfig
|
|
18
|
+
|
|
19
|
+
CONFIG_NAME = "config.toml"
|
|
20
|
+
MANIFEST_NAME = "guardrails.toml"
|
|
21
|
+
STORE_DIR = "store"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass
|
|
25
|
+
class HomeConfig:
|
|
26
|
+
home: Path
|
|
27
|
+
mailbox: MailboxConfig | None
|
|
28
|
+
manifest_path: Path
|
|
29
|
+
store_path: Path
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def write_config(
|
|
33
|
+
home: Path,
|
|
34
|
+
*,
|
|
35
|
+
host: str,
|
|
36
|
+
user: str,
|
|
37
|
+
folder: str = "INBOX",
|
|
38
|
+
password: str | None = None,
|
|
39
|
+
) -> Path:
|
|
40
|
+
home.mkdir(parents=True, exist_ok=True)
|
|
41
|
+
cfg = home / CONFIG_NAME
|
|
42
|
+
lines = [
|
|
43
|
+
"[mailbox]",
|
|
44
|
+
f'host = "{host}"',
|
|
45
|
+
f'user = "{user}"',
|
|
46
|
+
f'folder = "{folder}"',
|
|
47
|
+
]
|
|
48
|
+
if password:
|
|
49
|
+
lines.append(f'password = "{password}"')
|
|
50
|
+
lines.append("")
|
|
51
|
+
cfg.write_text("\n".join(lines))
|
|
52
|
+
os.chmod(cfg, stat.S_IRUSR | stat.S_IWUSR)
|
|
53
|
+
return cfg
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def load_home(home: str | Path) -> HomeConfig:
|
|
57
|
+
home = Path(home)
|
|
58
|
+
cfg_path = home / CONFIG_NAME
|
|
59
|
+
mailbox = None
|
|
60
|
+
if cfg_path.exists():
|
|
61
|
+
data = tomllib.loads(cfg_path.read_text())
|
|
62
|
+
mb = data.get("mailbox", {})
|
|
63
|
+
password = os.environ.get("DOCKETRY_IMAP_PASSWORD") or mb.get("password", "")
|
|
64
|
+
if mb.get("host") and mb.get("user"):
|
|
65
|
+
mailbox = MailboxConfig(
|
|
66
|
+
host=mb["host"],
|
|
67
|
+
user=mb["user"],
|
|
68
|
+
password=password,
|
|
69
|
+
folder=mb.get("folder", "INBOX"),
|
|
70
|
+
port=int(mb.get("port", 993)),
|
|
71
|
+
)
|
|
72
|
+
return HomeConfig(
|
|
73
|
+
home=home,
|
|
74
|
+
mailbox=mailbox,
|
|
75
|
+
manifest_path=home / MANIFEST_NAME,
|
|
76
|
+
store_path=home / STORE_DIR,
|
|
77
|
+
)
|
docketry/envelope.py
ADDED
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
"""MIME message -> normalized Envelope.
|
|
2
|
+
|
|
3
|
+
The port's one job: whatever arrives, reduce it to the same provenance-stamped
|
|
4
|
+
shape before anything downstream sees it. Parsing is stdlib-only and read-only;
|
|
5
|
+
the raw message is never modified, and the raw hash travels with the envelope
|
|
6
|
+
so provenance survives every later stage.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import email
|
|
11
|
+
import email.policy
|
|
12
|
+
import hashlib
|
|
13
|
+
import re
|
|
14
|
+
from dataclasses import asdict, dataclass, field
|
|
15
|
+
from email.message import EmailMessage
|
|
16
|
+
from email.utils import getaddresses, parsedate_to_datetime
|
|
17
|
+
from html.parser import HTMLParser
|
|
18
|
+
|
|
19
|
+
_FILENAME_SAFE = re.compile(r"[^A-Za-z0-9._ ()\[\]-]")
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class _HTMLText(HTMLParser):
|
|
23
|
+
_SKIP = {"script", "style", "head"}
|
|
24
|
+
_BREAK = {"p", "br", "div", "tr", "li", "table"}
|
|
25
|
+
|
|
26
|
+
def __init__(self) -> None:
|
|
27
|
+
super().__init__()
|
|
28
|
+
self._chunks: list[str] = []
|
|
29
|
+
self._skip_depth = 0
|
|
30
|
+
|
|
31
|
+
def handle_starttag(self, tag: str, attrs) -> None:
|
|
32
|
+
if tag in self._SKIP:
|
|
33
|
+
self._skip_depth += 1
|
|
34
|
+
elif tag in self._BREAK:
|
|
35
|
+
self._chunks.append("\n")
|
|
36
|
+
|
|
37
|
+
def handle_endtag(self, tag: str) -> None:
|
|
38
|
+
if tag in self._SKIP and self._skip_depth:
|
|
39
|
+
self._skip_depth -= 1
|
|
40
|
+
|
|
41
|
+
def handle_data(self, data: str) -> None:
|
|
42
|
+
if not self._skip_depth:
|
|
43
|
+
self._chunks.append(data)
|
|
44
|
+
|
|
45
|
+
def text(self) -> str:
|
|
46
|
+
raw = "".join(self._chunks)
|
|
47
|
+
lines = [ln.strip() for ln in raw.splitlines()]
|
|
48
|
+
return "\n".join(ln for ln in lines if ln)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def html_to_text(html: str) -> str:
|
|
52
|
+
parser = _HTMLText()
|
|
53
|
+
try:
|
|
54
|
+
parser.feed(html)
|
|
55
|
+
parser.close()
|
|
56
|
+
except Exception:
|
|
57
|
+
return html
|
|
58
|
+
return parser.text()
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def sanitize_filename(name: str) -> str:
|
|
62
|
+
"""Keep only the basename and drop anything shell- or path-hostile."""
|
|
63
|
+
name = name.replace("\\", "/").split("/")[-1].strip()
|
|
64
|
+
name = _FILENAME_SAFE.sub("_", name)
|
|
65
|
+
return name or "attachment.bin"
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
@dataclass
|
|
69
|
+
class Attachment:
|
|
70
|
+
filename: str
|
|
71
|
+
content_type: str
|
|
72
|
+
sha256: str
|
|
73
|
+
size: int
|
|
74
|
+
content: bytes = field(repr=False, compare=False)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
@dataclass
|
|
78
|
+
class Envelope:
|
|
79
|
+
message_id: str
|
|
80
|
+
from_addr: str
|
|
81
|
+
to: list[str]
|
|
82
|
+
cc: list[str]
|
|
83
|
+
date: str # ISO 8601, "" when unparseable
|
|
84
|
+
subject: str
|
|
85
|
+
body_text: str
|
|
86
|
+
attachments: list[Attachment]
|
|
87
|
+
raw_sha256: str
|
|
88
|
+
source: str # intake mailbox this arrived through
|
|
89
|
+
fetched_at: str # ISO 8601, stamped by the port
|
|
90
|
+
|
|
91
|
+
def to_record(self) -> dict:
|
|
92
|
+
"""JSON-safe form; attachment bytes are stored on disk, not in the row."""
|
|
93
|
+
d = asdict(self)
|
|
94
|
+
for a in d["attachments"]:
|
|
95
|
+
a.pop("content", None)
|
|
96
|
+
return d
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _addresses(msg: EmailMessage, header: str) -> list[str]:
|
|
100
|
+
return [addr for _, addr in getaddresses(msg.get_all(header, [])) if addr]
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _body_text(msg: EmailMessage) -> str:
|
|
104
|
+
body = msg.get_body(preferencelist=("plain", "html"))
|
|
105
|
+
if body is None:
|
|
106
|
+
return ""
|
|
107
|
+
try:
|
|
108
|
+
content = body.get_content()
|
|
109
|
+
except Exception:
|
|
110
|
+
payload = body.get_payload(decode=True) or b""
|
|
111
|
+
content = payload.decode("utf-8", "replace")
|
|
112
|
+
if body.get_content_type() == "text/html":
|
|
113
|
+
return html_to_text(content)
|
|
114
|
+
return content.strip()
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _attachments(msg: EmailMessage) -> list[Attachment]:
|
|
118
|
+
out: list[Attachment] = []
|
|
119
|
+
for part in msg.iter_attachments():
|
|
120
|
+
content = part.get_payload(decode=True)
|
|
121
|
+
if content is None:
|
|
122
|
+
payload = part.get_payload()
|
|
123
|
+
content = payload.encode("utf-8", "replace") if isinstance(payload, str) else b""
|
|
124
|
+
out.append(
|
|
125
|
+
Attachment(
|
|
126
|
+
filename=sanitize_filename(part.get_filename() or "attachment.bin"),
|
|
127
|
+
content_type=part.get_content_type(),
|
|
128
|
+
sha256=hashlib.sha256(content).hexdigest(),
|
|
129
|
+
size=len(content),
|
|
130
|
+
content=content,
|
|
131
|
+
)
|
|
132
|
+
)
|
|
133
|
+
return out
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def parse_message(raw: bytes, *, source: str, fetched_at: str) -> Envelope:
|
|
137
|
+
msg = email.message_from_bytes(raw, policy=email.policy.default)
|
|
138
|
+
raw_sha = hashlib.sha256(raw).hexdigest()
|
|
139
|
+
|
|
140
|
+
message_id = (msg.get("Message-ID") or "").strip().strip("<>")
|
|
141
|
+
if not message_id:
|
|
142
|
+
message_id = f"docketry-{raw_sha[:32]}"
|
|
143
|
+
|
|
144
|
+
date_iso = ""
|
|
145
|
+
if msg.get("Date"):
|
|
146
|
+
try:
|
|
147
|
+
date_iso = parsedate_to_datetime(msg["Date"]).isoformat()
|
|
148
|
+
except Exception:
|
|
149
|
+
date_iso = ""
|
|
150
|
+
|
|
151
|
+
from_pairs = getaddresses(msg.get_all("From", []))
|
|
152
|
+
from_addr = from_pairs[0][1] if from_pairs else ""
|
|
153
|
+
|
|
154
|
+
return Envelope(
|
|
155
|
+
message_id=message_id,
|
|
156
|
+
from_addr=from_addr,
|
|
157
|
+
to=_addresses(msg, "To"),
|
|
158
|
+
cc=_addresses(msg, "Cc"),
|
|
159
|
+
date=date_iso,
|
|
160
|
+
subject=str(msg.get("Subject") or ""),
|
|
161
|
+
body_text=_body_text(msg),
|
|
162
|
+
attachments=_attachments(msg),
|
|
163
|
+
raw_sha256=raw_sha,
|
|
164
|
+
source=source,
|
|
165
|
+
fetched_at=fetched_at,
|
|
166
|
+
)
|
docketry/extract.py
ADDED
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
"""Text extraction layer: attachment in, text with a page map out.
|
|
2
|
+
|
|
3
|
+
One interface for every downstream tool (classifier, citation verifier,
|
|
4
|
+
linter). Provenance and confidence travel with the text:
|
|
5
|
+
|
|
6
|
+
- PDF: native text per page via pypdf (extra: pdf). Pages with no native
|
|
7
|
+
text are reported as warnings; when OCR is requested/available the whole
|
|
8
|
+
scanned document is OCR'd page by page and every OCR page carries a mean
|
|
9
|
+
word confidence (0-100).
|
|
10
|
+
- OCR: requires the Tesseract binary + poppler's pdftoppm on the system, and
|
|
11
|
+
the pytesseract/Pillow packages (extra: ocr). Missing pieces raise
|
|
12
|
+
ExtractionError naming exactly what to install — low-confidence text is
|
|
13
|
+
flagged, garbage is never passed downstream silently.
|
|
14
|
+
- DOCX: python-docx paragraphs + table text (extra: docx). Word documents
|
|
15
|
+
have no fixed pages, so the result is one logical page and a warning says
|
|
16
|
+
pin-citing to page numbers is not possible from this format.
|
|
17
|
+
- TXT: stdlib.
|
|
18
|
+
"""
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import shutil
|
|
22
|
+
import subprocess
|
|
23
|
+
import tempfile
|
|
24
|
+
from dataclasses import dataclass, field
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
|
|
27
|
+
LOW_CONFIDENCE = 60.0
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class ExtractionError(RuntimeError):
|
|
31
|
+
pass
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class Page:
|
|
36
|
+
number: int
|
|
37
|
+
text: str
|
|
38
|
+
method: str # native | ocr | docx | text
|
|
39
|
+
confidence: float | None = None # OCR mean word confidence, None otherwise
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class Extraction:
|
|
44
|
+
pages: list[Page]
|
|
45
|
+
method: str
|
|
46
|
+
warnings: list[str] = field(default_factory=list)
|
|
47
|
+
|
|
48
|
+
@property
|
|
49
|
+
def full_text(self) -> str:
|
|
50
|
+
return "\n".join(p.text for p in self.pages)
|
|
51
|
+
|
|
52
|
+
def page_for_offset(self, offset: int) -> int | None:
|
|
53
|
+
"""Map a character offset in full_text back to a page number."""
|
|
54
|
+
pos = 0
|
|
55
|
+
for p in self.pages:
|
|
56
|
+
end = pos + len(p.text)
|
|
57
|
+
if offset <= end:
|
|
58
|
+
return p.number
|
|
59
|
+
pos = end + 1 # the joining newline
|
|
60
|
+
return self.pages[-1].number if self.pages else None
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _require(module: str, extra: str):
|
|
64
|
+
try:
|
|
65
|
+
return __import__(module)
|
|
66
|
+
except ImportError:
|
|
67
|
+
raise ExtractionError(
|
|
68
|
+
f"extracting this file type needs the '{extra}' extra:"
|
|
69
|
+
f" pip install 'docketry[{extra}]'"
|
|
70
|
+
) from None
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _extract_pdf(path: Path, *, ocr: str) -> Extraction:
|
|
74
|
+
pypdf = _require("pypdf", "pdf")
|
|
75
|
+
reader = pypdf.PdfReader(str(path))
|
|
76
|
+
pages: list[Page] = []
|
|
77
|
+
warnings: list[str] = []
|
|
78
|
+
empty = 0
|
|
79
|
+
for i, page in enumerate(reader.pages, start=1):
|
|
80
|
+
text = (page.extract_text() or "").strip()
|
|
81
|
+
if not text:
|
|
82
|
+
empty += 1
|
|
83
|
+
warnings.append(f"page {i}: no extractable text (likely scanned)")
|
|
84
|
+
pages.append(Page(number=i, text=text, method="native"))
|
|
85
|
+
|
|
86
|
+
mostly_empty = empty > len(pages) / 2 if pages else False
|
|
87
|
+
if ocr == "always" or (ocr == "auto" and mostly_empty):
|
|
88
|
+
try:
|
|
89
|
+
return _ocr_pdf(path)
|
|
90
|
+
except ExtractionError:
|
|
91
|
+
if ocr == "always":
|
|
92
|
+
raise
|
|
93
|
+
warnings.append(
|
|
94
|
+
f"{empty}/{len(pages)} pages have no native text and OCR is"
|
|
95
|
+
" unavailable — install the 'ocr' extra plus tesseract and"
|
|
96
|
+
" poppler-utils for scanned documents"
|
|
97
|
+
)
|
|
98
|
+
return Extraction(pages=pages, method="native", warnings=warnings)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _ocr_pdf(path: Path) -> Extraction:
|
|
102
|
+
pytesseract = _require("pytesseract", "ocr")
|
|
103
|
+
_require("PIL", "ocr")
|
|
104
|
+
from PIL import Image
|
|
105
|
+
|
|
106
|
+
for binary, package in (("tesseract", "tesseract-ocr"), ("pdftoppm", "poppler-utils")):
|
|
107
|
+
if shutil.which(binary) is None:
|
|
108
|
+
raise ExtractionError(
|
|
109
|
+
f"OCR needs the '{binary}' system binary (install {package})"
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
pages: list[Page] = []
|
|
113
|
+
warnings: list[str] = []
|
|
114
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
115
|
+
subprocess.run(
|
|
116
|
+
["pdftoppm", "-r", "300", "-png", str(path), f"{tmp}/page"],
|
|
117
|
+
check=True, capture_output=True,
|
|
118
|
+
)
|
|
119
|
+
images = sorted(Path(tmp).glob("page*.png"))
|
|
120
|
+
if not images:
|
|
121
|
+
raise ExtractionError("pdftoppm produced no page images")
|
|
122
|
+
for i, img_path in enumerate(images, start=1):
|
|
123
|
+
with Image.open(img_path) as img:
|
|
124
|
+
data = pytesseract.image_to_data(
|
|
125
|
+
img, output_type=pytesseract.Output.DICT
|
|
126
|
+
)
|
|
127
|
+
words, confs = [], []
|
|
128
|
+
for word, conf in zip(data["text"], data["conf"]):
|
|
129
|
+
if word.strip() and float(conf) >= 0:
|
|
130
|
+
words.append(word)
|
|
131
|
+
confs.append(float(conf))
|
|
132
|
+
confidence = sum(confs) / len(confs) if confs else 0.0
|
|
133
|
+
if confidence < LOW_CONFIDENCE:
|
|
134
|
+
warnings.append(
|
|
135
|
+
f"page {i}: OCR confidence {confidence:.0f} is below"
|
|
136
|
+
f" {LOW_CONFIDENCE:.0f} — text may be unreliable"
|
|
137
|
+
)
|
|
138
|
+
pages.append(
|
|
139
|
+
Page(number=i, text=" ".join(words), method="ocr", confidence=confidence)
|
|
140
|
+
)
|
|
141
|
+
return Extraction(pages=pages, method="ocr", warnings=warnings)
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _extract_docx(path: Path) -> Extraction:
|
|
145
|
+
docx = _require("docx", "docx")
|
|
146
|
+
document = docx.Document(str(path))
|
|
147
|
+
parts = [p.text for p in document.paragraphs if p.text.strip()]
|
|
148
|
+
for table in document.tables:
|
|
149
|
+
for row in table.rows:
|
|
150
|
+
cells = [c.text.strip() for c in row.cells if c.text.strip()]
|
|
151
|
+
if cells:
|
|
152
|
+
parts.append(" | ".join(cells))
|
|
153
|
+
return Extraction(
|
|
154
|
+
pages=[Page(number=1, text="\n".join(parts), method="docx")],
|
|
155
|
+
method="docx",
|
|
156
|
+
warnings=["DOCX has no fixed pages; page-level pin cites are not"
|
|
157
|
+
" derivable from this format"],
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _extract_txt(path: Path) -> Extraction:
|
|
162
|
+
text = path.read_bytes().decode("utf-8", "replace")
|
|
163
|
+
return Extraction(pages=[Page(number=1, text=text, method="text")], method="text")
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
_DISPATCH = {
|
|
167
|
+
".pdf": _extract_pdf,
|
|
168
|
+
".docx": _extract_docx,
|
|
169
|
+
".txt": _extract_txt,
|
|
170
|
+
".md": _extract_txt,
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def extract_path(path: str | Path, *, ocr: str = "auto") -> Extraction:
|
|
175
|
+
"""Extract text from a file. ocr: "auto" | "always" | "never"."""
|
|
176
|
+
if ocr not in ("auto", "always", "never"):
|
|
177
|
+
raise ValueError(f"ocr must be auto/always/never, not {ocr!r}")
|
|
178
|
+
path = Path(path)
|
|
179
|
+
if not path.exists():
|
|
180
|
+
raise ExtractionError(f"no such file: {path}")
|
|
181
|
+
handler = _DISPATCH.get(path.suffix.lower())
|
|
182
|
+
if handler is None:
|
|
183
|
+
raise ExtractionError(
|
|
184
|
+
f"unsupported file type '{path.suffix}' (supported:"
|
|
185
|
+
f" {', '.join(sorted(_DISPATCH))})"
|
|
186
|
+
)
|
|
187
|
+
if handler is _extract_pdf:
|
|
188
|
+
if ocr == "never":
|
|
189
|
+
return _extract_pdf_no_ocr(path)
|
|
190
|
+
return handler(path, ocr=ocr)
|
|
191
|
+
return handler(path)
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def _extract_pdf_no_ocr(path: Path) -> Extraction:
|
|
195
|
+
return _extract_pdf(path, ocr="off")
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"""Gate registry: manifests reference gates by id; plugins register here."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
_REGISTRY: dict[str, type] = {}
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def register(cls: type) -> type:
|
|
8
|
+
gate_id = getattr(cls, "id", None)
|
|
9
|
+
if not gate_id:
|
|
10
|
+
raise ValueError(f"{cls.__name__} has no id")
|
|
11
|
+
_REGISTRY[gate_id] = cls
|
|
12
|
+
return cls
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def get(gate_id: str) -> type:
|
|
16
|
+
if gate_id not in _REGISTRY:
|
|
17
|
+
raise KeyError(
|
|
18
|
+
f"unknown gate '{gate_id}' (registered: {', '.join(sorted(_REGISTRY)) or 'none'})"
|
|
19
|
+
)
|
|
20
|
+
return _REGISTRY[gate_id]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def all_ids() -> list[str]:
|
|
24
|
+
return sorted(_REGISTRY)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
# Built-ins register on import.
|
|
28
|
+
from . import builtin, classifier, notice # noqa: E402,F401
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
"""Starter gates. Deterministic pipeline-hygiene checks only.
|
|
2
|
+
|
|
3
|
+
None of these is a security control. Docketry makes no malware, phishing, or
|
|
4
|
+
other cybersecurity claims anywhere — these gates decide what the *pipeline*
|
|
5
|
+
will accept, nothing more. Firms should get actual security from their mail
|
|
6
|
+
provider and endpoint tooling.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from ..envelope import Envelope
|
|
11
|
+
from ..pipeline import Finding, SEVERITY_FAIL, SEVERITY_INFO
|
|
12
|
+
from . import register
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@register
|
|
16
|
+
class AttachmentPolicy:
|
|
17
|
+
"""What file types and sizes this pipeline accepts. Hygiene, not AV."""
|
|
18
|
+
|
|
19
|
+
id = "attachment-policy"
|
|
20
|
+
allowed_stages = {"ingest"}
|
|
21
|
+
|
|
22
|
+
_DEFAULT_DENY = (
|
|
23
|
+
".exe .js .vbs .scr .bat .cmd .com .ps1 .jar .msi .hta .lnk".split()
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
def validate_options(self, options: dict) -> list[str]:
|
|
27
|
+
problems = []
|
|
28
|
+
if "max_size_mb" in options:
|
|
29
|
+
try:
|
|
30
|
+
float(options["max_size_mb"])
|
|
31
|
+
except (TypeError, ValueError):
|
|
32
|
+
problems.append("max_size_mb must be a number")
|
|
33
|
+
deny = options.get("deny_extensions")
|
|
34
|
+
if deny is not None and (not isinstance(deny, list)
|
|
35
|
+
or not all(isinstance(e, str) for e in deny)):
|
|
36
|
+
problems.append("deny_extensions must be a list of strings")
|
|
37
|
+
return problems
|
|
38
|
+
|
|
39
|
+
def check(self, env: Envelope, options: dict) -> list[Finding]:
|
|
40
|
+
deny = {e.lower() for e in options.get("deny_extensions", self._DEFAULT_DENY)}
|
|
41
|
+
max_mb = float(options.get("max_size_mb", 25))
|
|
42
|
+
findings: list[Finding] = []
|
|
43
|
+
for a in env.attachments:
|
|
44
|
+
ext = ("." + a.filename.rsplit(".", 1)[-1].lower()) if "." in a.filename else ""
|
|
45
|
+
if ext in deny:
|
|
46
|
+
findings.append(
|
|
47
|
+
Finding(self.id, SEVERITY_FAIL, f"attachment type not accepted: {a.filename}")
|
|
48
|
+
)
|
|
49
|
+
if a.size > max_mb * 1024 * 1024:
|
|
50
|
+
findings.append(
|
|
51
|
+
Finding(
|
|
52
|
+
self.id,
|
|
53
|
+
SEVERITY_FAIL,
|
|
54
|
+
f"attachment over {max_mb:g} MB: {a.filename} ({a.size} bytes)",
|
|
55
|
+
)
|
|
56
|
+
)
|
|
57
|
+
return findings
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@register
|
|
61
|
+
class SenderScope:
|
|
62
|
+
"""Optionally hold mail from senders outside the expected set.
|
|
63
|
+
|
|
64
|
+
An intake mailbox fed by forwarding rules mostly hears from known portals
|
|
65
|
+
and staff; anything else bounces to a human instead of flowing onward.
|
|
66
|
+
"""
|
|
67
|
+
|
|
68
|
+
id = "sender-scope"
|
|
69
|
+
allowed_stages = {"ingest"}
|
|
70
|
+
|
|
71
|
+
def validate_options(self, options: dict) -> list[str]:
|
|
72
|
+
problems = []
|
|
73
|
+
for key in ("allow", "deny"):
|
|
74
|
+
val = options.get(key)
|
|
75
|
+
if val is not None and (not isinstance(val, list)
|
|
76
|
+
or not all(isinstance(s, str) for s in val)):
|
|
77
|
+
problems.append(f"{key} must be a list of strings")
|
|
78
|
+
return problems
|
|
79
|
+
|
|
80
|
+
@staticmethod
|
|
81
|
+
def _matches(sender: str, domain: str, entry: str) -> bool:
|
|
82
|
+
if sender == entry:
|
|
83
|
+
return True
|
|
84
|
+
if entry.startswith("@"):
|
|
85
|
+
entry_domain = entry[1:]
|
|
86
|
+
# "@uscourts.gov" covers the domain and its subdomains —
|
|
87
|
+
# federal NEFs arrive from per-district hosts like
|
|
88
|
+
# flsd.uscourts.gov.
|
|
89
|
+
return domain == entry_domain or domain.endswith("." + entry_domain)
|
|
90
|
+
return False
|
|
91
|
+
|
|
92
|
+
def check(self, env: Envelope, options: dict) -> list[Finding]:
|
|
93
|
+
sender = env.from_addr.lower()
|
|
94
|
+
domain = sender.rsplit("@", 1)[-1] if "@" in sender else ""
|
|
95
|
+
for entry in (s.lower() for s in options.get("deny", [])):
|
|
96
|
+
if self._matches(sender, domain, entry):
|
|
97
|
+
return [Finding(self.id, SEVERITY_FAIL,
|
|
98
|
+
f"sender on the deny list: {env.from_addr}")]
|
|
99
|
+
allowed = [s.lower() for s in options.get("allow", [])]
|
|
100
|
+
if not allowed:
|
|
101
|
+
return []
|
|
102
|
+
for entry in allowed:
|
|
103
|
+
if self._matches(sender, domain, entry):
|
|
104
|
+
return []
|
|
105
|
+
return [
|
|
106
|
+
Finding(self.id, SEVERITY_FAIL, f"sender outside intake scope: {env.from_addr}")
|
|
107
|
+
]
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
@register
|
|
111
|
+
class ProvenanceStamp:
|
|
112
|
+
"""Records where each message came from; informational, never holds."""
|
|
113
|
+
|
|
114
|
+
id = "provenance-stamp"
|
|
115
|
+
allowed_stages = None
|
|
116
|
+
|
|
117
|
+
def check(self, env: Envelope, options: dict) -> list[Finding]:
|
|
118
|
+
return [
|
|
119
|
+
Finding(
|
|
120
|
+
self.id,
|
|
121
|
+
SEVERITY_INFO,
|
|
122
|
+
f"source={env.source} raw_sha256={env.raw_sha256[:16]} fetched_at={env.fetched_at}",
|
|
123
|
+
)
|
|
124
|
+
]
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
@register
|
|
128
|
+
class NameScreen:
|
|
129
|
+
"""Hold any message whose content mentions a screened name.
|
|
130
|
+
|
|
131
|
+
The legal use is an ethical wall / conflict screen: a firm lists the
|
|
132
|
+
parties or matters a reviewer must not see flowing through the pipeline
|
|
133
|
+
unreviewed, and anything mentioning them parks for the declared
|
|
134
|
+
authority. Terms match case-insensitively on word boundaries across
|
|
135
|
+
subject, body, and attachment filenames. The screen list lives in the
|
|
136
|
+
firm's local guardrails.toml and never leaves the machine.
|
|
137
|
+
"""
|
|
138
|
+
|
|
139
|
+
id = "name-screen"
|
|
140
|
+
allowed_stages = None
|
|
141
|
+
|
|
142
|
+
def validate_options(self, options: dict) -> list[str]:
|
|
143
|
+
terms = options.get("terms")
|
|
144
|
+
if not isinstance(terms, list) or not terms or not all(
|
|
145
|
+
isinstance(t, str) and t.strip() for t in terms
|
|
146
|
+
):
|
|
147
|
+
return ["terms must be a non-empty list of strings"]
|
|
148
|
+
return []
|
|
149
|
+
|
|
150
|
+
def check(self, env: Envelope, options: dict) -> list[Finding]:
|
|
151
|
+
import re as _re
|
|
152
|
+
|
|
153
|
+
haystack = "\n".join(
|
|
154
|
+
[env.subject, env.body_text, *[a.filename for a in env.attachments]]
|
|
155
|
+
)
|
|
156
|
+
note = options.get("note", "screened name")
|
|
157
|
+
findings = []
|
|
158
|
+
for term in options.get("terms", []):
|
|
159
|
+
if _re.search(rf"\b{_re.escape(term)}\b", haystack, _re.IGNORECASE):
|
|
160
|
+
findings.append(Finding(
|
|
161
|
+
self.id, SEVERITY_FAIL,
|
|
162
|
+
f"message mentions '{term}' ({note}) — held for the declared authority",
|
|
163
|
+
))
|
|
164
|
+
return findings
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"""Doc-classifier gate: proposes a type for each attachment, never applies.
|
|
2
|
+
|
|
3
|
+
Informational only — the write path is the staged queue (class-apply),
|
|
4
|
+
fill-only and role-recorded. Classification is free and deterministic first;
|
|
5
|
+
model tiers are somebody else's optional add-on, never a default.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from ..classify import classify
|
|
10
|
+
from ..envelope import Envelope
|
|
11
|
+
from ..pipeline import Finding, SEVERITY_INFO
|
|
12
|
+
from . import register
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@register
|
|
16
|
+
class DocClassifier:
|
|
17
|
+
id = "doc-classifier"
|
|
18
|
+
allowed_stages = None
|
|
19
|
+
|
|
20
|
+
def check(self, env: Envelope, options: dict) -> list[Finding]:
|
|
21
|
+
findings = []
|
|
22
|
+
for a in env.attachments:
|
|
23
|
+
label, tier = classify(a.filename)
|
|
24
|
+
if tier != "low":
|
|
25
|
+
findings.append(
|
|
26
|
+
Finding(self.id, SEVERITY_INFO, f"{a.filename}: proposed {label} ({tier})")
|
|
27
|
+
)
|
|
28
|
+
return findings
|