docketry 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
docketry/config.py ADDED
@@ -0,0 +1,77 @@
1
+ """Home-directory config for a Docketry installation.
2
+
3
+ A Docketry "home" is one directory holding config.toml, the guardrail
4
+ manifest, the SQLite store, and attachments — the whole installation is one
5
+ folder on the firm's own disk. The IMAP password is read from the
6
+ DOCKETRY_IMAP_PASSWORD environment variable first; storing it in config.toml
7
+ is supported for single-machine setups (the file is chmod 0600 on write).
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import os
12
+ import stat
13
+ import tomllib
14
+ from dataclasses import dataclass
15
+ from pathlib import Path
16
+
17
+ from .mailbox import MailboxConfig
18
+
19
+ CONFIG_NAME = "config.toml"
20
+ MANIFEST_NAME = "guardrails.toml"
21
+ STORE_DIR = "store"
22
+
23
+
24
+ @dataclass
25
+ class HomeConfig:
26
+ home: Path
27
+ mailbox: MailboxConfig | None
28
+ manifest_path: Path
29
+ store_path: Path
30
+
31
+
32
+ def write_config(
33
+ home: Path,
34
+ *,
35
+ host: str,
36
+ user: str,
37
+ folder: str = "INBOX",
38
+ password: str | None = None,
39
+ ) -> Path:
40
+ home.mkdir(parents=True, exist_ok=True)
41
+ cfg = home / CONFIG_NAME
42
+ lines = [
43
+ "[mailbox]",
44
+ f'host = "{host}"',
45
+ f'user = "{user}"',
46
+ f'folder = "{folder}"',
47
+ ]
48
+ if password:
49
+ lines.append(f'password = "{password}"')
50
+ lines.append("")
51
+ cfg.write_text("\n".join(lines))
52
+ os.chmod(cfg, stat.S_IRUSR | stat.S_IWUSR)
53
+ return cfg
54
+
55
+
56
+ def load_home(home: str | Path) -> HomeConfig:
57
+ home = Path(home)
58
+ cfg_path = home / CONFIG_NAME
59
+ mailbox = None
60
+ if cfg_path.exists():
61
+ data = tomllib.loads(cfg_path.read_text())
62
+ mb = data.get("mailbox", {})
63
+ password = os.environ.get("DOCKETRY_IMAP_PASSWORD") or mb.get("password", "")
64
+ if mb.get("host") and mb.get("user"):
65
+ mailbox = MailboxConfig(
66
+ host=mb["host"],
67
+ user=mb["user"],
68
+ password=password,
69
+ folder=mb.get("folder", "INBOX"),
70
+ port=int(mb.get("port", 993)),
71
+ )
72
+ return HomeConfig(
73
+ home=home,
74
+ mailbox=mailbox,
75
+ manifest_path=home / MANIFEST_NAME,
76
+ store_path=home / STORE_DIR,
77
+ )
docketry/envelope.py ADDED
@@ -0,0 +1,166 @@
1
+ """MIME message -> normalized Envelope.
2
+
3
+ The port's one job: whatever arrives, reduce it to the same provenance-stamped
4
+ shape before anything downstream sees it. Parsing is stdlib-only and read-only;
5
+ the raw message is never modified, and the raw hash travels with the envelope
6
+ so provenance survives every later stage.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import email
11
+ import email.policy
12
+ import hashlib
13
+ import re
14
+ from dataclasses import asdict, dataclass, field
15
+ from email.message import EmailMessage
16
+ from email.utils import getaddresses, parsedate_to_datetime
17
+ from html.parser import HTMLParser
18
+
19
+ _FILENAME_SAFE = re.compile(r"[^A-Za-z0-9._ ()\[\]-]")
20
+
21
+
22
+ class _HTMLText(HTMLParser):
23
+ _SKIP = {"script", "style", "head"}
24
+ _BREAK = {"p", "br", "div", "tr", "li", "table"}
25
+
26
+ def __init__(self) -> None:
27
+ super().__init__()
28
+ self._chunks: list[str] = []
29
+ self._skip_depth = 0
30
+
31
+ def handle_starttag(self, tag: str, attrs) -> None:
32
+ if tag in self._SKIP:
33
+ self._skip_depth += 1
34
+ elif tag in self._BREAK:
35
+ self._chunks.append("\n")
36
+
37
+ def handle_endtag(self, tag: str) -> None:
38
+ if tag in self._SKIP and self._skip_depth:
39
+ self._skip_depth -= 1
40
+
41
+ def handle_data(self, data: str) -> None:
42
+ if not self._skip_depth:
43
+ self._chunks.append(data)
44
+
45
+ def text(self) -> str:
46
+ raw = "".join(self._chunks)
47
+ lines = [ln.strip() for ln in raw.splitlines()]
48
+ return "\n".join(ln for ln in lines if ln)
49
+
50
+
51
+ def html_to_text(html: str) -> str:
52
+ parser = _HTMLText()
53
+ try:
54
+ parser.feed(html)
55
+ parser.close()
56
+ except Exception:
57
+ return html
58
+ return parser.text()
59
+
60
+
61
+ def sanitize_filename(name: str) -> str:
62
+ """Keep only the basename and drop anything shell- or path-hostile."""
63
+ name = name.replace("\\", "/").split("/")[-1].strip()
64
+ name = _FILENAME_SAFE.sub("_", name)
65
+ return name or "attachment.bin"
66
+
67
+
68
+ @dataclass
69
+ class Attachment:
70
+ filename: str
71
+ content_type: str
72
+ sha256: str
73
+ size: int
74
+ content: bytes = field(repr=False, compare=False)
75
+
76
+
77
+ @dataclass
78
+ class Envelope:
79
+ message_id: str
80
+ from_addr: str
81
+ to: list[str]
82
+ cc: list[str]
83
+ date: str # ISO 8601, "" when unparseable
84
+ subject: str
85
+ body_text: str
86
+ attachments: list[Attachment]
87
+ raw_sha256: str
88
+ source: str # intake mailbox this arrived through
89
+ fetched_at: str # ISO 8601, stamped by the port
90
+
91
+ def to_record(self) -> dict:
92
+ """JSON-safe form; attachment bytes are stored on disk, not in the row."""
93
+ d = asdict(self)
94
+ for a in d["attachments"]:
95
+ a.pop("content", None)
96
+ return d
97
+
98
+
99
+ def _addresses(msg: EmailMessage, header: str) -> list[str]:
100
+ return [addr for _, addr in getaddresses(msg.get_all(header, [])) if addr]
101
+
102
+
103
+ def _body_text(msg: EmailMessage) -> str:
104
+ body = msg.get_body(preferencelist=("plain", "html"))
105
+ if body is None:
106
+ return ""
107
+ try:
108
+ content = body.get_content()
109
+ except Exception:
110
+ payload = body.get_payload(decode=True) or b""
111
+ content = payload.decode("utf-8", "replace")
112
+ if body.get_content_type() == "text/html":
113
+ return html_to_text(content)
114
+ return content.strip()
115
+
116
+
117
+ def _attachments(msg: EmailMessage) -> list[Attachment]:
118
+ out: list[Attachment] = []
119
+ for part in msg.iter_attachments():
120
+ content = part.get_payload(decode=True)
121
+ if content is None:
122
+ payload = part.get_payload()
123
+ content = payload.encode("utf-8", "replace") if isinstance(payload, str) else b""
124
+ out.append(
125
+ Attachment(
126
+ filename=sanitize_filename(part.get_filename() or "attachment.bin"),
127
+ content_type=part.get_content_type(),
128
+ sha256=hashlib.sha256(content).hexdigest(),
129
+ size=len(content),
130
+ content=content,
131
+ )
132
+ )
133
+ return out
134
+
135
+
136
+ def parse_message(raw: bytes, *, source: str, fetched_at: str) -> Envelope:
137
+ msg = email.message_from_bytes(raw, policy=email.policy.default)
138
+ raw_sha = hashlib.sha256(raw).hexdigest()
139
+
140
+ message_id = (msg.get("Message-ID") or "").strip().strip("<>")
141
+ if not message_id:
142
+ message_id = f"docketry-{raw_sha[:32]}"
143
+
144
+ date_iso = ""
145
+ if msg.get("Date"):
146
+ try:
147
+ date_iso = parsedate_to_datetime(msg["Date"]).isoformat()
148
+ except Exception:
149
+ date_iso = ""
150
+
151
+ from_pairs = getaddresses(msg.get_all("From", []))
152
+ from_addr = from_pairs[0][1] if from_pairs else ""
153
+
154
+ return Envelope(
155
+ message_id=message_id,
156
+ from_addr=from_addr,
157
+ to=_addresses(msg, "To"),
158
+ cc=_addresses(msg, "Cc"),
159
+ date=date_iso,
160
+ subject=str(msg.get("Subject") or ""),
161
+ body_text=_body_text(msg),
162
+ attachments=_attachments(msg),
163
+ raw_sha256=raw_sha,
164
+ source=source,
165
+ fetched_at=fetched_at,
166
+ )
docketry/extract.py ADDED
@@ -0,0 +1,195 @@
1
+ """Text extraction layer: attachment in, text with a page map out.
2
+
3
+ One interface for every downstream tool (classifier, citation verifier,
4
+ linter). Provenance and confidence travel with the text:
5
+
6
+ - PDF: native text per page via pypdf (extra: pdf). Pages with no native
7
+ text are reported as warnings; when OCR is requested/available the whole
8
+ scanned document is OCR'd page by page and every OCR page carries a mean
9
+ word confidence (0-100).
10
+ - OCR: requires the Tesseract binary + poppler's pdftoppm on the system, and
11
+ the pytesseract/Pillow packages (extra: ocr). Missing pieces raise
12
+ ExtractionError naming exactly what to install — low-confidence text is
13
+ flagged, garbage is never passed downstream silently.
14
+ - DOCX: python-docx paragraphs + table text (extra: docx). Word documents
15
+ have no fixed pages, so the result is one logical page and a warning says
16
+ pin-citing to page numbers is not possible from this format.
17
+ - TXT: stdlib.
18
+ """
19
+ from __future__ import annotations
20
+
21
+ import shutil
22
+ import subprocess
23
+ import tempfile
24
+ from dataclasses import dataclass, field
25
+ from pathlib import Path
26
+
27
+ LOW_CONFIDENCE = 60.0
28
+
29
+
30
+ class ExtractionError(RuntimeError):
31
+ pass
32
+
33
+
34
+ @dataclass
35
+ class Page:
36
+ number: int
37
+ text: str
38
+ method: str # native | ocr | docx | text
39
+ confidence: float | None = None # OCR mean word confidence, None otherwise
40
+
41
+
42
+ @dataclass
43
+ class Extraction:
44
+ pages: list[Page]
45
+ method: str
46
+ warnings: list[str] = field(default_factory=list)
47
+
48
+ @property
49
+ def full_text(self) -> str:
50
+ return "\n".join(p.text for p in self.pages)
51
+
52
+ def page_for_offset(self, offset: int) -> int | None:
53
+ """Map a character offset in full_text back to a page number."""
54
+ pos = 0
55
+ for p in self.pages:
56
+ end = pos + len(p.text)
57
+ if offset <= end:
58
+ return p.number
59
+ pos = end + 1 # the joining newline
60
+ return self.pages[-1].number if self.pages else None
61
+
62
+
63
+ def _require(module: str, extra: str):
64
+ try:
65
+ return __import__(module)
66
+ except ImportError:
67
+ raise ExtractionError(
68
+ f"extracting this file type needs the '{extra}' extra:"
69
+ f" pip install 'docketry[{extra}]'"
70
+ ) from None
71
+
72
+
73
+ def _extract_pdf(path: Path, *, ocr: str) -> Extraction:
74
+ pypdf = _require("pypdf", "pdf")
75
+ reader = pypdf.PdfReader(str(path))
76
+ pages: list[Page] = []
77
+ warnings: list[str] = []
78
+ empty = 0
79
+ for i, page in enumerate(reader.pages, start=1):
80
+ text = (page.extract_text() or "").strip()
81
+ if not text:
82
+ empty += 1
83
+ warnings.append(f"page {i}: no extractable text (likely scanned)")
84
+ pages.append(Page(number=i, text=text, method="native"))
85
+
86
+ mostly_empty = empty > len(pages) / 2 if pages else False
87
+ if ocr == "always" or (ocr == "auto" and mostly_empty):
88
+ try:
89
+ return _ocr_pdf(path)
90
+ except ExtractionError:
91
+ if ocr == "always":
92
+ raise
93
+ warnings.append(
94
+ f"{empty}/{len(pages)} pages have no native text and OCR is"
95
+ " unavailable — install the 'ocr' extra plus tesseract and"
96
+ " poppler-utils for scanned documents"
97
+ )
98
+ return Extraction(pages=pages, method="native", warnings=warnings)
99
+
100
+
101
+ def _ocr_pdf(path: Path) -> Extraction:
102
+ pytesseract = _require("pytesseract", "ocr")
103
+ _require("PIL", "ocr")
104
+ from PIL import Image
105
+
106
+ for binary, package in (("tesseract", "tesseract-ocr"), ("pdftoppm", "poppler-utils")):
107
+ if shutil.which(binary) is None:
108
+ raise ExtractionError(
109
+ f"OCR needs the '{binary}' system binary (install {package})"
110
+ )
111
+
112
+ pages: list[Page] = []
113
+ warnings: list[str] = []
114
+ with tempfile.TemporaryDirectory() as tmp:
115
+ subprocess.run(
116
+ ["pdftoppm", "-r", "300", "-png", str(path), f"{tmp}/page"],
117
+ check=True, capture_output=True,
118
+ )
119
+ images = sorted(Path(tmp).glob("page*.png"))
120
+ if not images:
121
+ raise ExtractionError("pdftoppm produced no page images")
122
+ for i, img_path in enumerate(images, start=1):
123
+ with Image.open(img_path) as img:
124
+ data = pytesseract.image_to_data(
125
+ img, output_type=pytesseract.Output.DICT
126
+ )
127
+ words, confs = [], []
128
+ for word, conf in zip(data["text"], data["conf"]):
129
+ if word.strip() and float(conf) >= 0:
130
+ words.append(word)
131
+ confs.append(float(conf))
132
+ confidence = sum(confs) / len(confs) if confs else 0.0
133
+ if confidence < LOW_CONFIDENCE:
134
+ warnings.append(
135
+ f"page {i}: OCR confidence {confidence:.0f} is below"
136
+ f" {LOW_CONFIDENCE:.0f} — text may be unreliable"
137
+ )
138
+ pages.append(
139
+ Page(number=i, text=" ".join(words), method="ocr", confidence=confidence)
140
+ )
141
+ return Extraction(pages=pages, method="ocr", warnings=warnings)
142
+
143
+
144
+ def _extract_docx(path: Path) -> Extraction:
145
+ docx = _require("docx", "docx")
146
+ document = docx.Document(str(path))
147
+ parts = [p.text for p in document.paragraphs if p.text.strip()]
148
+ for table in document.tables:
149
+ for row in table.rows:
150
+ cells = [c.text.strip() for c in row.cells if c.text.strip()]
151
+ if cells:
152
+ parts.append(" | ".join(cells))
153
+ return Extraction(
154
+ pages=[Page(number=1, text="\n".join(parts), method="docx")],
155
+ method="docx",
156
+ warnings=["DOCX has no fixed pages; page-level pin cites are not"
157
+ " derivable from this format"],
158
+ )
159
+
160
+
161
+ def _extract_txt(path: Path) -> Extraction:
162
+ text = path.read_bytes().decode("utf-8", "replace")
163
+ return Extraction(pages=[Page(number=1, text=text, method="text")], method="text")
164
+
165
+
166
+ _DISPATCH = {
167
+ ".pdf": _extract_pdf,
168
+ ".docx": _extract_docx,
169
+ ".txt": _extract_txt,
170
+ ".md": _extract_txt,
171
+ }
172
+
173
+
174
+ def extract_path(path: str | Path, *, ocr: str = "auto") -> Extraction:
175
+ """Extract text from a file. ocr: "auto" | "always" | "never"."""
176
+ if ocr not in ("auto", "always", "never"):
177
+ raise ValueError(f"ocr must be auto/always/never, not {ocr!r}")
178
+ path = Path(path)
179
+ if not path.exists():
180
+ raise ExtractionError(f"no such file: {path}")
181
+ handler = _DISPATCH.get(path.suffix.lower())
182
+ if handler is None:
183
+ raise ExtractionError(
184
+ f"unsupported file type '{path.suffix}' (supported:"
185
+ f" {', '.join(sorted(_DISPATCH))})"
186
+ )
187
+ if handler is _extract_pdf:
188
+ if ocr == "never":
189
+ return _extract_pdf_no_ocr(path)
190
+ return handler(path, ocr=ocr)
191
+ return handler(path)
192
+
193
+
194
+ def _extract_pdf_no_ocr(path: Path) -> Extraction:
195
+ return _extract_pdf(path, ocr="off")
@@ -0,0 +1,28 @@
1
+ """Gate registry: manifests reference gates by id; plugins register here."""
2
+ from __future__ import annotations
3
+
4
+ _REGISTRY: dict[str, type] = {}
5
+
6
+
7
+ def register(cls: type) -> type:
8
+ gate_id = getattr(cls, "id", None)
9
+ if not gate_id:
10
+ raise ValueError(f"{cls.__name__} has no id")
11
+ _REGISTRY[gate_id] = cls
12
+ return cls
13
+
14
+
15
+ def get(gate_id: str) -> type:
16
+ if gate_id not in _REGISTRY:
17
+ raise KeyError(
18
+ f"unknown gate '{gate_id}' (registered: {', '.join(sorted(_REGISTRY)) or 'none'})"
19
+ )
20
+ return _REGISTRY[gate_id]
21
+
22
+
23
+ def all_ids() -> list[str]:
24
+ return sorted(_REGISTRY)
25
+
26
+
27
+ # Built-ins register on import.
28
+ from . import builtin, classifier, notice # noqa: E402,F401
@@ -0,0 +1,164 @@
1
+ """Starter gates. Deterministic pipeline-hygiene checks only.
2
+
3
+ None of these is a security control. Docketry makes no malware, phishing, or
4
+ other cybersecurity claims anywhere — these gates decide what the *pipeline*
5
+ will accept, nothing more. Firms should get actual security from their mail
6
+ provider and endpoint tooling.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ from ..envelope import Envelope
11
+ from ..pipeline import Finding, SEVERITY_FAIL, SEVERITY_INFO
12
+ from . import register
13
+
14
+
15
+ @register
16
+ class AttachmentPolicy:
17
+ """What file types and sizes this pipeline accepts. Hygiene, not AV."""
18
+
19
+ id = "attachment-policy"
20
+ allowed_stages = {"ingest"}
21
+
22
+ _DEFAULT_DENY = (
23
+ ".exe .js .vbs .scr .bat .cmd .com .ps1 .jar .msi .hta .lnk".split()
24
+ )
25
+
26
+ def validate_options(self, options: dict) -> list[str]:
27
+ problems = []
28
+ if "max_size_mb" in options:
29
+ try:
30
+ float(options["max_size_mb"])
31
+ except (TypeError, ValueError):
32
+ problems.append("max_size_mb must be a number")
33
+ deny = options.get("deny_extensions")
34
+ if deny is not None and (not isinstance(deny, list)
35
+ or not all(isinstance(e, str) for e in deny)):
36
+ problems.append("deny_extensions must be a list of strings")
37
+ return problems
38
+
39
+ def check(self, env: Envelope, options: dict) -> list[Finding]:
40
+ deny = {e.lower() for e in options.get("deny_extensions", self._DEFAULT_DENY)}
41
+ max_mb = float(options.get("max_size_mb", 25))
42
+ findings: list[Finding] = []
43
+ for a in env.attachments:
44
+ ext = ("." + a.filename.rsplit(".", 1)[-1].lower()) if "." in a.filename else ""
45
+ if ext in deny:
46
+ findings.append(
47
+ Finding(self.id, SEVERITY_FAIL, f"attachment type not accepted: {a.filename}")
48
+ )
49
+ if a.size > max_mb * 1024 * 1024:
50
+ findings.append(
51
+ Finding(
52
+ self.id,
53
+ SEVERITY_FAIL,
54
+ f"attachment over {max_mb:g} MB: {a.filename} ({a.size} bytes)",
55
+ )
56
+ )
57
+ return findings
58
+
59
+
60
+ @register
61
+ class SenderScope:
62
+ """Optionally hold mail from senders outside the expected set.
63
+
64
+ An intake mailbox fed by forwarding rules mostly hears from known portals
65
+ and staff; anything else bounces to a human instead of flowing onward.
66
+ """
67
+
68
+ id = "sender-scope"
69
+ allowed_stages = {"ingest"}
70
+
71
+ def validate_options(self, options: dict) -> list[str]:
72
+ problems = []
73
+ for key in ("allow", "deny"):
74
+ val = options.get(key)
75
+ if val is not None and (not isinstance(val, list)
76
+ or not all(isinstance(s, str) for s in val)):
77
+ problems.append(f"{key} must be a list of strings")
78
+ return problems
79
+
80
+ @staticmethod
81
+ def _matches(sender: str, domain: str, entry: str) -> bool:
82
+ if sender == entry:
83
+ return True
84
+ if entry.startswith("@"):
85
+ entry_domain = entry[1:]
86
+ # "@uscourts.gov" covers the domain and its subdomains —
87
+ # federal NEFs arrive from per-district hosts like
88
+ # flsd.uscourts.gov.
89
+ return domain == entry_domain or domain.endswith("." + entry_domain)
90
+ return False
91
+
92
+ def check(self, env: Envelope, options: dict) -> list[Finding]:
93
+ sender = env.from_addr.lower()
94
+ domain = sender.rsplit("@", 1)[-1] if "@" in sender else ""
95
+ for entry in (s.lower() for s in options.get("deny", [])):
96
+ if self._matches(sender, domain, entry):
97
+ return [Finding(self.id, SEVERITY_FAIL,
98
+ f"sender on the deny list: {env.from_addr}")]
99
+ allowed = [s.lower() for s in options.get("allow", [])]
100
+ if not allowed:
101
+ return []
102
+ for entry in allowed:
103
+ if self._matches(sender, domain, entry):
104
+ return []
105
+ return [
106
+ Finding(self.id, SEVERITY_FAIL, f"sender outside intake scope: {env.from_addr}")
107
+ ]
108
+
109
+
110
+ @register
111
+ class ProvenanceStamp:
112
+ """Records where each message came from; informational, never holds."""
113
+
114
+ id = "provenance-stamp"
115
+ allowed_stages = None
116
+
117
+ def check(self, env: Envelope, options: dict) -> list[Finding]:
118
+ return [
119
+ Finding(
120
+ self.id,
121
+ SEVERITY_INFO,
122
+ f"source={env.source} raw_sha256={env.raw_sha256[:16]} fetched_at={env.fetched_at}",
123
+ )
124
+ ]
125
+
126
+
127
+ @register
128
+ class NameScreen:
129
+ """Hold any message whose content mentions a screened name.
130
+
131
+ The legal use is an ethical wall / conflict screen: a firm lists the
132
+ parties or matters a reviewer must not see flowing through the pipeline
133
+ unreviewed, and anything mentioning them parks for the declared
134
+ authority. Terms match case-insensitively on word boundaries across
135
+ subject, body, and attachment filenames. The screen list lives in the
136
+ firm's local guardrails.toml and never leaves the machine.
137
+ """
138
+
139
+ id = "name-screen"
140
+ allowed_stages = None
141
+
142
+ def validate_options(self, options: dict) -> list[str]:
143
+ terms = options.get("terms")
144
+ if not isinstance(terms, list) or not terms or not all(
145
+ isinstance(t, str) and t.strip() for t in terms
146
+ ):
147
+ return ["terms must be a non-empty list of strings"]
148
+ return []
149
+
150
+ def check(self, env: Envelope, options: dict) -> list[Finding]:
151
+ import re as _re
152
+
153
+ haystack = "\n".join(
154
+ [env.subject, env.body_text, *[a.filename for a in env.attachments]]
155
+ )
156
+ note = options.get("note", "screened name")
157
+ findings = []
158
+ for term in options.get("terms", []):
159
+ if _re.search(rf"\b{_re.escape(term)}\b", haystack, _re.IGNORECASE):
160
+ findings.append(Finding(
161
+ self.id, SEVERITY_FAIL,
162
+ f"message mentions '{term}' ({note}) — held for the declared authority",
163
+ ))
164
+ return findings
@@ -0,0 +1,28 @@
1
+ """Doc-classifier gate: proposes a type for each attachment, never applies.
2
+
3
+ Informational only — the write path is the staged queue (class-apply),
4
+ fill-only and role-recorded. Classification is free and deterministic first;
5
+ model tiers are somebody else's optional add-on, never a default.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ from ..classify import classify
10
+ from ..envelope import Envelope
11
+ from ..pipeline import Finding, SEVERITY_INFO
12
+ from . import register
13
+
14
+
15
+ @register
16
+ class DocClassifier:
17
+ id = "doc-classifier"
18
+ allowed_stages = None
19
+
20
+ def check(self, env: Envelope, options: dict) -> list[Finding]:
21
+ findings = []
22
+ for a in env.attachments:
23
+ label, tier = classify(a.filename)
24
+ if tier != "low":
25
+ findings.append(
26
+ Finding(self.id, SEVERITY_INFO, f"{a.filename}: proposed {label} ({tier})")
27
+ )
28
+ return findings