openom-core 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,65 @@
1
+ # SPDX-License-Identifier: MIT
2
+ """openOM deterministic core.
3
+
4
+ Embed / read / inspect / validate machine-readable, broker-asserted payloads in
5
+ commercial-real-estate offering-memorandum PDFs. Zero inference, ever (CLAUDE.md Rule 1).
6
+
7
+ The stable public surface is re-exported here, so an integrator writes
8
+ ``from openom_core import embed, read, validate, load_schema`` (submodule paths keep working too).
9
+ See the spec Part II §A-§E, §H-§J.
10
+ """
11
+
12
+ from importlib.metadata import PackageNotFoundError
13
+ from importlib.metadata import version as _pkg_version
14
+
15
+ from .canonical import canonicalize, hash_bytes, payload_hash
16
+ from .embed import ReadResult, embed, read, reembed_warnings
17
+ from .errors import CanonicalizationError, Finding, PayloadTooLargeError, Severity
18
+ from .images import ImageManifest, extract_images
19
+ from .inspect import Profile, classify
20
+ from .schema import load_schema
21
+ from .summary import DealSummary, summarize_deal
22
+ from .text import TextResult, extract_text
23
+ from .types import OMPayload, RealEstateListing
24
+ from .validate import Report, Tolerances, validate
25
+
26
+ try: # single source of truth: the installed package metadata (matches `pip show` / the wheel)
27
+ __version__ = _pkg_version("openom-core")
28
+ except PackageNotFoundError: # editable/source tree without an installed dist
29
+ __version__ = "0.0.0+unknown"
30
+
31
+ #: The openOM spec version this library targets. Drift-locked to the schema by tests/test_types.py.
32
+ SPEC_VERSION = "0.1"
33
+
34
+ __all__ = [
35
+ "SPEC_VERSION",
36
+ "__version__",
37
+ # verbs
38
+ "embed",
39
+ "read",
40
+ "reembed_warnings",
41
+ "validate",
42
+ "classify",
43
+ "extract_text",
44
+ "extract_images",
45
+ "canonicalize",
46
+ "hash_bytes",
47
+ "payload_hash",
48
+ "load_schema",
49
+ "summarize_deal",
50
+ # result / option types
51
+ "ReadResult",
52
+ "Report",
53
+ "Tolerances",
54
+ "Profile",
55
+ "TextResult",
56
+ "ImageManifest",
57
+ "DealSummary",
58
+ "Finding",
59
+ "Severity",
60
+ "OMPayload",
61
+ "RealEstateListing",
62
+ # errors
63
+ "CanonicalizationError",
64
+ "PayloadTooLargeError",
65
+ ]
@@ -0,0 +1,166 @@
1
+ # SPDX-License-Identifier: MIT
2
+ """RFC 8785 JSON Canonicalization (JCS) + the openOM integrity hash (spec §C).
3
+
4
+ The keystone of cross-implementation fidelity: two conformant implementations MUST produce
5
+ byte-identical output here, and therefore the same SHA-256. RFC 8785 itself performs no
6
+ Unicode normalization and assumes unique member names (RFC 8785 §3.1), so §C.1 mandates the
7
+ preprocessing done here (NFC, duplicate-key rejection, number-range checks) *before* JCS.
8
+ Serialization (key sorting, minimal escaping, ES number formatting) is delegated to the
9
+ vetted ``rfc8785`` library.
10
+
11
+ Producer vs Consumer contract (§C, §D)
12
+ --------------------------------------
13
+ The **producer** normalizes (NFC), canonicalizes, hashes the resulting bytes, and stores
14
+ *those exact bytes* as the payload plus the hash in the XMP marker. The **consumer** hashes
15
+ the stored bytes *as received* - it does NOT re-canonicalize before verifying. Verification
16
+ is therefore a byte comparison of ``sha256(stored_bytes)`` against the marker hash; it never
17
+ depends on the consumer re-running NFC/JCS. This asymmetry is deliberate: normalization
18
+ happens once, at authoring time, so a consumer on a different platform/library cannot perturb
19
+ the hash. ``canonicalize`` is the producer path; ``hash_bytes`` over stored bytes is the
20
+ consumer path (see :func:`openom_core.embed.read`).
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ import copy
26
+ import hashlib
27
+ import json
28
+ import math
29
+ from collections.abc import Mapping
30
+ from typing import Any
31
+ from unicodedata import normalize
32
+
33
+ import rfc8785
34
+
35
+ from .errors import IO_BADUTF8, IO_DUPKEY, IO_NUMRANGE, IO_STRUCTURE, CanonicalizationError
36
+
37
+ #: ECMAScript safe-integer limit; integers beyond this are silently rounded by the number
38
+ #: model, which would be data corruption (§C [OM-CANON-013]).
39
+ MAX_SAFE_INT = 2**53 - 1
40
+
41
+ #: Max nesting depth (§J JSON-hardening guard) - matches the JS parser for cross-impl parity.
42
+ MAX_DEPTH = 64
43
+
44
+
45
+ def _prepare(obj: Any, depth: int = 0) -> Any:
46
+ """NFC-normalize strings + member names, reject duplicate keys and non-representable numbers.
47
+
48
+ Returns a new structure ready for RFC 8785 serialization. Mutates nothing.
49
+ """
50
+ if depth > MAX_DEPTH:
51
+ raise CanonicalizationError(IO_STRUCTURE, f"nesting exceeds {MAX_DEPTH}")
52
+ # bool is an int subclass - must be checked first.
53
+ if isinstance(obj, bool):
54
+ return obj
55
+ if isinstance(obj, str):
56
+ # A lone UTF-16 surrogate cannot be encoded as UTF-8; reject it explicitly with a
57
+ # stable code rather than letting the serializer raise a bare UnicodeEncodeError.
58
+ if any(0xD800 <= ord(ch) <= 0xDFFF for ch in obj):
59
+ raise CanonicalizationError(IO_BADUTF8, "string contains an unpaired surrogate")
60
+ return normalize("NFC", obj)
61
+ if isinstance(obj, Mapping):
62
+ out: dict[str, Any] = {}
63
+ for key, value in obj.items():
64
+ if not isinstance(key, str):
65
+ raise CanonicalizationError(IO_DUPKEY, f"non-string member name: {key!r}")
66
+ nkey = normalize("NFC", key)
67
+ if nkey in out:
68
+ raise CanonicalizationError(IO_DUPKEY, f"duplicate member name after NFC: {nkey!r}")
69
+ out[nkey] = _prepare(value, depth + 1)
70
+ return out
71
+ if isinstance(obj, (list, tuple)):
72
+ return [_prepare(item, depth + 1) for item in obj]
73
+ if isinstance(obj, int):
74
+ if abs(obj) > MAX_SAFE_INT:
75
+ raise CanonicalizationError(IO_NUMRANGE, f"integer exceeds 2^53-1: {obj}")
76
+ return obj
77
+ if isinstance(obj, float):
78
+ if not math.isfinite(obj):
79
+ raise CanonicalizationError(IO_NUMRANGE, f"non-finite number: {obj}")
80
+ if obj.is_integer() and abs(obj) > MAX_SAFE_INT:
81
+ raise CanonicalizationError(IO_NUMRANGE, f"float integer exceeds 2^53-1: {obj}")
82
+ return obj
83
+ if obj is None:
84
+ return None
85
+ raise CanonicalizationError(IO_NUMRANGE, f"unsupported JSON type: {type(obj).__name__}")
86
+
87
+
88
+ def _reject_duplicate_pairs(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
89
+ """json object_pairs_hook: reject duplicate member names (after NFC) rather than last-wins.
90
+ Preserves the ORIGINAL keys (does not normalize) so a reader returns the payload as stored."""
91
+ out: dict[str, Any] = {}
92
+ seen: set[str] = set()
93
+ for key, value in pairs:
94
+ nkey = normalize("NFC", key)
95
+ if nkey in seen:
96
+ raise CanonicalizationError(IO_DUPKEY, f"duplicate member name after NFC: {nkey!r}")
97
+ seen.add(nkey)
98
+ out[key] = value
99
+ return out
100
+
101
+
102
+ def _check_depth(obj: Any, depth: int = 0) -> None:
103
+ if depth > MAX_DEPTH:
104
+ raise CanonicalizationError(IO_STRUCTURE, f"nesting exceeds {MAX_DEPTH}")
105
+ if isinstance(obj, Mapping):
106
+ for v in obj.values():
107
+ _check_depth(v, depth + 1)
108
+ elif isinstance(obj, (list, tuple)):
109
+ for v in obj:
110
+ _check_depth(v, depth + 1)
111
+
112
+
113
+ def parse_hardened(raw: bytes | str) -> Any:
114
+ """Parse an om.json payload with the §J read-side hardening the write path enforces
115
+ ([OM-CANON-009/010]): reject duplicate member names and over-deep nesting. Used by the reusable
116
+ ``read`` verb so a self-hoster calling core on untrusted PDFs gets the MCP invariants."""
117
+ obj = json.loads(raw, object_pairs_hook=_reject_duplicate_pairs)
118
+ _check_depth(obj)
119
+ return obj
120
+
121
+
122
+ def canonicalize(payload: Mapping[str, Any]) -> bytes:
123
+ """Serialize a payload to its RFC 8785 JCS bytes (UTF-8, no BOM). Producer path.
124
+
125
+ Applies the §C.1 preprocessing (NFC, unique keys, number range) then delegates
126
+ serialization to ``rfc8785``. The top level MUST be a JSON object (§C.10); a bare
127
+ array/scalar is rejected rather than hashed.
128
+ """
129
+ if not isinstance(payload, Mapping):
130
+ raise CanonicalizationError(
131
+ IO_STRUCTURE, f"top-level value must be an object, got {type(payload).__name__}"
132
+ )
133
+ prepared = _prepare(payload)
134
+ try:
135
+ return rfc8785.dumps(prepared)
136
+ except CanonicalizationError:
137
+ raise
138
+ except Exception as exc: # noqa: BLE001 - normalize any serializer failure to our code
139
+ raise CanonicalizationError(IO_NUMRANGE, str(exc)) from exc
140
+
141
+
142
+ def hash_bytes(data: bytes) -> str:
143
+ """The openOM integrity hash of already-canonical bytes: ``sha256:<lowercase-hex>``.
144
+
145
+ Used on both the write path (over the bytes just produced) and the read/verify path
146
+ (over the decompressed stored bytes, as received - no re-canonicalization; §C, §D).
147
+ """
148
+ return "sha256:" + hashlib.sha256(data).hexdigest()
149
+
150
+
151
+ def strip_signature(payload: Mapping[str, Any]) -> dict[str, Any]:
152
+ """Return a deep copy with ``meta.signature`` *removed* (not nulled), per [OM-CANON-003].
153
+
154
+ In 0.1 the signature is always absent/null, so this is a no-op on real payloads; it exists
155
+ so that adding a signature in a future version does not change the integrity hash.
156
+ """
157
+ out = copy.deepcopy(dict(payload))
158
+ meta = out.get("meta")
159
+ if isinstance(meta, dict) and "signature" in meta:
160
+ meta.pop("signature", None)
161
+ return out
162
+
163
+
164
+ def payload_hash(payload: Mapping[str, Any]) -> str:
165
+ """Convenience: the integrity hash of a payload object (strip signature → JCS → sha256)."""
166
+ return hash_bytes(canonicalize(strip_signature(payload)))
openom_core/embed.py ADDED
@@ -0,0 +1,437 @@
1
+ # SPDX-License-Identifier: MIT
2
+ """Embed / read the om.json payload in a PDF via pikepdf (spec §D).
3
+
4
+ Non-destructive by construction: pikepdf appends the embedded file + XMP marker without
5
+ touching page content. The catalog ``/AF`` array is added manually - assigning to
6
+ ``Pdf.attachments`` populates the ``/EmbeddedFiles`` name tree but NOT ``/AF`` ([OM-EMB-002]).
7
+ The exact JCS bytes are stored verbatim ([OM-EMB-010]); the integrity hash is over those bytes.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import io
13
+ import os
14
+ import re
15
+ import tempfile
16
+ import zlib
17
+ from dataclasses import dataclass
18
+ from typing import Any
19
+
20
+ import pikepdf
21
+
22
+ from .canonical import canonicalize, hash_bytes, parse_hardened, strip_signature
23
+ from .errors import Finding, PayloadTooLargeError, SignedEmbedError
24
+ from .xmp import _marker_props, read_marker, render_marker_xml, write_marker
25
+
26
+ #: Decompressed-payload cap (§J [OM-SEC-002]).
27
+ MAX_PAYLOAD_BYTES = 5_000_000
28
+
29
+ PAYLOAD_NAME = "om.json"
30
+ MIME = "application/ld+json"
31
+ SUBTYPE = pikepdf.Name("/application/ld+json") # serialized name-escaped as /application#2Fld+json
32
+
33
+
34
+ @dataclass
35
+ class ReadResult:
36
+ """Result of reading a payload from a PDF (§I om_read shape)."""
37
+
38
+ present: bool
39
+ payload: dict[str, Any] | None
40
+ hash_valid: bool | None
41
+ origin_verified: None = None # read-time origin check is a Consumer concern; null in core
42
+ signature_valid: None = None # reserved (§10 layer 4)
43
+ source_doc_hash: str | None = None # #5: marker sourceDocHash (provenance of the source PDF)
44
+
45
+
46
+ def _remove_existing(pdf: pikepdf.Pdf) -> None:
47
+ """Remove any existing om.json attachment and its /AF reference (idempotent embed)."""
48
+ if PAYLOAD_NAME not in pdf.attachments:
49
+ return
50
+ old_objgen = pdf.attachments[PAYLOAD_NAME].obj.objgen
51
+ # Filter /AF while the referenced object is still live, then delete the attachment.
52
+ if "/AF" in pdf.Root:
53
+ kept = [f for f in pdf.Root.AF if f.objgen != old_objgen]
54
+ if kept:
55
+ pdf.Root.AF = pikepdf.Array(kept)
56
+ else:
57
+ del pdf.Root.AF
58
+ del pdf.attachments[PAYLOAD_NAME]
59
+
60
+
61
+ def _ensure_af(pdf: pikepdf.Pdf, spec: pikepdf.Object) -> None:
62
+ """Ensure the catalog /AF array references the payload's Filespec ([OM-EMB-002])."""
63
+ if "/AF" not in pdf.Root:
64
+ pdf.Root.AF = pikepdf.Array([spec])
65
+ return
66
+ if all(f.objgen != spec.objgen for f in pdf.Root.AF):
67
+ pdf.Root.AF.append(spec)
68
+
69
+
70
+ def _set_subtype(spec: pikepdf.Object) -> None:
71
+ """Set /Subtype application/ld+json on the embedded-file stream(s) ([OM-EMB-004])."""
72
+ ef = spec.EF
73
+ for key in ("/F", "/UF"):
74
+ if key in ef:
75
+ ef[key].Subtype = SUBTYPE
76
+
77
+
78
+ def input_encrypted(pdf_bytes: bytes) -> bool:
79
+ """True if the input PDF is encrypted (permission encryption with an empty user password, which
80
+ pikepdf opens transparently and embed() then writes out UNENCRYPTED - a silent security-posture
81
+ change worth signaling to the author). A password-protected PDF raises on open and never reaches
82
+ here, so this is specifically the restrictions-only case (#4)."""
83
+ try:
84
+ with pikepdf.open(io.BytesIO(pdf_bytes)) as pdf:
85
+ return bool(pdf.is_encrypted)
86
+ except pikepdf.PasswordError:
87
+ return False
88
+
89
+
90
+ def _is_signed(pdf: pikepdf.Pdf) -> bool:
91
+ """True if the PDF carries a digital signature (§10 layer 4 / #3 [OM-EMB-020]).
92
+
93
+ A full-rewrite save invalidates a byte-range signature; when this is True, embed uses an
94
+ incremental-update save instead, appending the payload so the signed bytes stay untouched.
95
+ Detects the AcroForm ``SigFlags`` "signatures exist" bit, any ``/FT /Sig`` field carrying a
96
+ ``/V``, or a ``/Perms`` (DocMDP/UR) entry - covering approval and certification signatures.
97
+ """
98
+ root = pdf.Root
99
+ if "/Perms" in root:
100
+ return True
101
+ acro = root.get("/AcroForm")
102
+ if acro is None:
103
+ return False
104
+ sig_flags = acro.get("/SigFlags")
105
+ if sig_flags is not None and int(sig_flags) & 1:
106
+ return True
107
+ fields = acro.get("/Fields")
108
+ if fields is None:
109
+ return False
110
+ return any(f.get("/FT") == pikepdf.Name.Sig and "/V" in f for f in fields)
111
+
112
+
113
+ @dataclass
114
+ class _EmbedFields:
115
+ data: bytes
116
+ payload_hash: str
117
+ spec_version: str
118
+ supersedes: str | None
119
+ source_doc_hash: str
120
+
121
+
122
+ def _plan_embed(pdf: pikepdf.Pdf, pdf_bytes: bytes, payload: dict[str, Any]) -> _EmbedFields:
123
+ """Compute the payload bytes + marker fields (supersedes/sourceDocHash) - shared by both the
124
+ full-rewrite and incremental save paths so a signed OM gets identical provenance semantics."""
125
+ # signature excluded from the integrity preimage ([OM-CANON-003])
126
+ data = canonicalize(strip_signature(payload))
127
+ if len(data) > MAX_PAYLOAD_BYTES:
128
+ raise PayloadTooLargeError(len(data), MAX_PAYLOAD_BYTES)
129
+ payload_hash = hash_bytes(data)
130
+ # §D.4 re-embed semantics:
131
+ # - a *different* prior payload is superseded (supersedes = prior hash);
132
+ # - an *identical* re-embed carries the prior supersedes forward (no lineage wipe, no
133
+ # self-supersede); a first embed has no predecessor (supersedes = None).
134
+ prior = read_marker(pdf)
135
+ prior_hash = prior.get("payloadHash") if prior else None
136
+ prior_supersedes = prior.get("supersedes") if prior else None
137
+ if prior_hash and prior_hash != payload_hash:
138
+ supersedes: str | None = prior_hash
139
+ elif prior_hash == payload_hash:
140
+ supersedes = prior_supersedes
141
+ else:
142
+ supersedes = None
143
+ # #5: sourceDocHash identifies the underlying source PDF, held STABLE across reprices - computed
144
+ # once (first embed) and carried forward from the prior marker on every re-embed.
145
+ prior_source = prior.get("sourceDocHash") if prior else None
146
+ source_doc_hash = prior_source or hash_bytes(pdf_bytes)
147
+ return _EmbedFields(
148
+ data=data,
149
+ payload_hash=payload_hash,
150
+ spec_version=str(payload.get("specVersion", "0.1")),
151
+ supersedes=supersedes,
152
+ source_doc_hash=source_doc_hash,
153
+ )
154
+
155
+
156
+ def embed(
157
+ pdf_bytes: bytes, payload: dict[str, Any], *, asserted_date: str, badge: bool = False
158
+ ) -> bytes:
159
+ """Embed ``payload`` as om.json and return the new PDF bytes. Never mutates the input.
160
+
161
+ A *signed* input (#3 [OM-EMB-020]) is embedded via an incremental-update save - the payload is
162
+ appended after the signed byte range so the signature stays cryptographically intact - rather
163
+ than the default full-rewrite (which would invalidate it). Both paths write the identical
164
+ payload bytes and XMP marker, so the result reads the same regardless of the save method.
165
+ """
166
+ with pikepdf.open(io.BytesIO(pdf_bytes)) as pdf:
167
+ fields = _plan_embed(pdf, pdf_bytes, payload)
168
+ signed = _is_signed(pdf)
169
+
170
+ if signed:
171
+ return _embed_incremental(pdf_bytes, fields, asserted_date)
172
+
173
+ with pikepdf.open(io.BytesIO(pdf_bytes)) as pdf:
174
+ _remove_existing(pdf)
175
+ # pikepdf's stub marks description/filename/dates as required; runtime defaults them.
176
+ # We intentionally omit dates for determinism (§D [OM-EMB-011]).
177
+ filespec = pikepdf.AttachedFileSpec(pdf, fields.data, mime_type=MIME) # type: ignore[call-arg]
178
+ filespec.relationship = pikepdf.Name.Data # /AFRelationship (kwarg missing from stub)
179
+ pdf.attachments[PAYLOAD_NAME] = filespec
180
+ spec_obj = pdf.attachments[PAYLOAD_NAME].obj
181
+ _ensure_af(pdf, spec_obj)
182
+ _set_subtype(spec_obj)
183
+ write_marker(
184
+ pdf,
185
+ spec_version=fields.spec_version,
186
+ payload_filename=PAYLOAD_NAME,
187
+ payload_hash=fields.payload_hash,
188
+ asserted_date=asserted_date,
189
+ supersedes=fields.supersedes,
190
+ source_doc_hash=fields.source_doc_hash,
191
+ )
192
+ out = io.BytesIO()
193
+ pdf.save(out, deterministic_id=True)
194
+ return out.getvalue()
195
+
196
+
197
+ def _embed_incremental(pdf_bytes: bytes, fields: _EmbedFields, asserted_date: str) -> bytes:
198
+ """Append the payload to a *signed* PDF via a fitz incremental-update save (#3 [OM-EMB-020]).
199
+
200
+ Builds the om.json embedded-file stream, an indirect /Filespec (/AFRelationship /Data,
201
+ /Subtype application/ld+json), the /EmbeddedFiles name-tree entry, the catalog /AF reference,
202
+ and the XMP marker - then saves incrementally so the original signed bytes are preserved as a
203
+ byte-exact prefix. The marker bytes come from the same ``render_marker_xml`` the full-rewrite
204
+ path uses, so the two producers agree. fitz's incremental save requires a real file whose bytes
205
+ equal the source, so the work happens in a temp file.
206
+ """
207
+ # PyMuPDF is the optional [render] extra; the signed-OM incremental path is the one embed case
208
+ # that needs it. Imported lazily with a clear hint so the pikepdf-only core stays MIT-clean.
209
+ try:
210
+ import pymupdf
211
+ except ImportError as exc:
212
+ raise SignedEmbedError(
213
+ "embedding a signed PDF needs PyMuPDF: pip install 'openom-core[render]'"
214
+ ) from exc
215
+
216
+ props = _marker_props(
217
+ spec_version=fields.spec_version,
218
+ payload_filename=PAYLOAD_NAME,
219
+ payload_hash=fields.payload_hash,
220
+ asserted_date=asserted_date,
221
+ supersedes=fields.supersedes,
222
+ source_doc_hash=fields.source_doc_hash,
223
+ )
224
+ # A TemporaryDirectory (vs NamedTemporaryFile + os.unlink) so cleanup can't mask the real error
225
+ # with a Windows "file in use" (WinError 32) in the finally block.
226
+ with tempfile.TemporaryDirectory(prefix="openom_signed_") as tmpdir:
227
+ path = os.path.join(tmpdir, "in.pdf")
228
+ with open(path, "wb") as fh:
229
+ fh.write(pdf_bytes)
230
+ doc = pymupdf.open(path)
231
+ try:
232
+ # If pymupdf had to REBUILD the xref to open this file, an incremental append is not a
233
+ # byte-exact extension of the original - the signed prefix would change and the
234
+ # signature break. Refuse cleanly (OM-EMB-021) rather than ship an invalid signature.
235
+ if getattr(doc, "is_repaired", False):
236
+ raise SignedEmbedError(
237
+ "this signed PDF needed its cross-reference table rebuilt to open, so an "
238
+ "incremental (signature-preserving) embed is not safe; embed an unsigned copy "
239
+ "or re-issue the signature after embedding"
240
+ )
241
+ cat = doc.pdf_catalog()
242
+ # 1. embedded-file stream (verbatim JCS bytes; pikepdf read decodes the filter)
243
+ ef = doc.get_new_xref()
244
+ doc.update_object(ef, "<< /Type /EmbeddedFile /Subtype /application#2Fld+json >>")
245
+ doc.update_stream(ef, fields.data, compress=True)
246
+ # 2. indirect /Filespec ([OM-EMB-002]/[OM-EMB-004])
247
+ spec = doc.get_new_xref()
248
+ doc.update_object(
249
+ spec,
250
+ f"<< /Type /Filespec /F ({PAYLOAD_NAME}) /UF ({PAYLOAD_NAME}) "
251
+ f"/AFRelationship /Data /Desc (openOM payload) "
252
+ f"/EF << /F {ef} 0 R /UF {ef} 0 R >> >>",
253
+ )
254
+ # 3. /EmbeddedFiles name tree - drop any prior om.json, keep other attachments
255
+ entries = [
256
+ f"({nm}) {xr} 0 R"
257
+ for nm, xr in _name_tree_entries(doc, cat)
258
+ if nm != PAYLOAD_NAME
259
+ ]
260
+ entries.append(f"({PAYLOAD_NAME}) {spec} 0 R")
261
+ doc.xref_set_key(cat, "Names/EmbeddedFiles/Names", "[" + " ".join(entries) + "]")
262
+ # 4. catalog /AF - drop any prior om.json filespec ref (by resolving each ref's /F,
263
+ # since a re-embed's prior filespec has a different xref), append ours
264
+ af = [r for r in _af_refs(doc, cat) if not _is_om_filespec(doc, r)]
265
+ af.append(f"{spec} 0 R")
266
+ doc.xref_set_key(cat, "AF", "[" + " ".join(af) + "]")
267
+ # 5. XMP marker (identical bytes to the full-rewrite path)
268
+ existing = _existing_metadata_xml(doc, cat)
269
+ meta = doc.get_new_xref()
270
+ doc.update_object(meta, "<< /Type /Metadata /Subtype /XML >>")
271
+ marker_xml = render_marker_xml(existing, props).encode("utf-8")
272
+ doc.update_stream(meta, marker_xml, compress=False)
273
+ doc.xref_set_key(cat, "Metadata", f"{meta} 0 R")
274
+ try:
275
+ doc.saveIncr()
276
+ except Exception as exc: # noqa: BLE001 - any fitz save failure -> typed refusal
277
+ raise SignedEmbedError(
278
+ f"incremental save failed on this signed PDF: {exc}"
279
+ ) from exc
280
+ finally:
281
+ doc.close()
282
+ with open(path, "rb") as fh:
283
+ out = fh.read()
284
+ # Postcondition the whole feature promises: the original signed bytes stay a byte-exact prefix,
285
+ # so every byte the /ByteRange signs is unchanged. Never return a doc that broke it.
286
+ if not out.startswith(pdf_bytes):
287
+ raise SignedEmbedError("incremental embed did not preserve the signed byte prefix")
288
+ return out
289
+
290
+
291
+ def _name_tree_entries(doc: Any, cat: int) -> list[tuple[str, str]]:
292
+ """(name, 'N 0 R') pairs already in /Names/EmbeddedFiles/Names, or [] if none."""
293
+ kind, val = doc.xref_get_key(cat, "Names/EmbeddedFiles/Names")
294
+ if kind != "array":
295
+ return []
296
+ return re.findall(r"\(([^)]*)\)\s*(\d+ 0 R)", val)
297
+
298
+
299
+ def _af_refs(doc: Any, cat: int) -> list[str]:
300
+ """Existing catalog /AF indirect refs ('N 0 R'), or [] if none."""
301
+ kind, val = doc.xref_get_key(cat, "AF")
302
+ if kind != "array":
303
+ return []
304
+ return re.findall(r"\d+ 0 R", val)
305
+
306
+
307
+ def _is_om_filespec(doc: Any, ref: str) -> bool:
308
+ """True if the /AF ref 'N 0 R' points to a filespec whose /F is our om.json (drop it on
309
+ re-embed so /AF never stacks duplicate payload references)."""
310
+ m = re.match(r"(\d+) 0 R", ref)
311
+ if not m:
312
+ return False
313
+ kind, val = doc.xref_get_key(int(m.group(1)), "F")
314
+ return bool(kind == "string" and val == PAYLOAD_NAME)
315
+
316
+
317
+ def _existing_metadata_xml(doc: Any, cat: int) -> str | None:
318
+ """The current catalog /Metadata XML, or None - so our marker merges alongside existing XMP."""
319
+ kind, val = doc.xref_get_key(cat, "Metadata")
320
+ if kind != "xref":
321
+ return None
322
+ meta_xref = int(re.match(r"(\d+) 0 R", val).group(1)) # type: ignore[union-attr]
323
+ raw = doc.xref_stream(meta_xref)
324
+ return raw.decode("utf-8", "replace") if raw else None
325
+
326
+
327
+ def reembed_warnings(
328
+ pdf_bytes: bytes, payload: dict[str, Any], *, asserted_date: str
329
+ ) -> list[Finding]:
330
+ """Non-blocking re-embed warnings against an existing PDF (§H). Pure; ``embed`` stays
331
+ bytes→bytes, so callers (CLI/MCP) compose this to surface provenance issues.
332
+
333
+ OMW-W051: the new assertedDate precedes the payload it would supersede (time going
334
+ backwards on a reprice). Warnings never block.
335
+ """
336
+ with pikepdf.open(io.BytesIO(pdf_bytes)) as pdf:
337
+ prior = read_marker(pdf)
338
+ if not prior:
339
+ return []
340
+ payload_hash = hash_bytes(canonicalize(strip_signature(payload)))
341
+ prior_hash = prior.get("payloadHash")
342
+ prior_date = prior.get("assertedDate")
343
+ out: list[Finding] = []
344
+ if prior_hash and prior_hash != payload_hash and prior_date and asserted_date < prior_date:
345
+ out.append(
346
+ Finding(
347
+ "OMW-W051",
348
+ "warning",
349
+ "/assertedDate",
350
+ "assertedDate precedes the superseded payload's assertedDate",
351
+ expected=prior_date,
352
+ actual=asserted_date,
353
+ )
354
+ )
355
+ return out
356
+
357
+
358
+ def _find_ef_stream(pdf: pikepdf.Pdf) -> pikepdf.Object | None:
359
+ """Locate the om.json embedded-file stream in spec detection order ([OM-XMP-003]).
360
+
361
+ Authoritative path: catalog ``/AF`` → Filespec (``/UF`` or ``/F`` == om.json) → ``/EF``.
362
+ Falls back to the ``/EmbeddedFiles`` name tree for producers that populate only that.
363
+ """
364
+
365
+ def _ef_of(spec: pikepdf.Object) -> pikepdf.Object | None:
366
+ if "/EF" not in spec:
367
+ return None
368
+ ef = spec.EF
369
+ for key in ("/UF", "/F"):
370
+ if key in ef:
371
+ return ef[key]
372
+ return None
373
+
374
+ if "/AF" in pdf.Root:
375
+ for spec in pdf.Root.AF:
376
+ names = {str(spec[k]) for k in ("/UF", "/F") if k in spec}
377
+ if PAYLOAD_NAME in names:
378
+ stream = _ef_of(spec)
379
+ if stream is not None:
380
+ return stream
381
+ if PAYLOAD_NAME in pdf.attachments: # fallback: EmbeddedFiles name tree
382
+ return _ef_of(pdf.attachments[PAYLOAD_NAME].obj)
383
+ return None
384
+
385
+
386
+ def _decoded_payload_bytes(stream: pikepdf.Object) -> bytes:
387
+ """Decode the EF stream to raw payload bytes, bounding size *before* full materialization.
388
+
389
+ A malicious PDF can hide a decompression bomb in a tiny compressed stream. We read the
390
+ stored (compressed) bytes - always bounded by the file itself - reject if already over the
391
+ cap, then inflate with a hard ceiling rather than decompressing unbounded into memory.
392
+ """
393
+ raw = bytes(stream.read_raw_bytes())
394
+ if len(raw) > MAX_PAYLOAD_BYTES:
395
+ raise PayloadTooLargeError(len(raw), MAX_PAYLOAD_BYTES)
396
+
397
+ filt = stream.get("/Filter")
398
+ names = [str(filt)] if isinstance(filt, pikepdf.Name) else [str(n) for n in (filt or [])]
399
+ # Plain FlateDecode (no predictor) - the common case for our writes and pdf-lib - can be
400
+ # inflated with a bounded zlib object. Anything exotic (predictors, other filters) defers
401
+ # to pikepdf's decoder, still guarded by the compressed-size cap above and a post-check.
402
+ if names == ["/FlateDecode"] and "/DecodeParms" not in stream and "/DP" not in stream:
403
+ data = _bounded_inflate(raw)
404
+ elif not names:
405
+ data = raw # stored uncompressed
406
+ else:
407
+ data = bytes(stream.read_bytes())
408
+ if len(data) > MAX_PAYLOAD_BYTES:
409
+ raise PayloadTooLargeError(len(data), MAX_PAYLOAD_BYTES)
410
+ return data
411
+
412
+
413
+ def _bounded_inflate(raw: bytes) -> bytes:
414
+ obj = zlib.decompressobj()
415
+ out = obj.decompress(raw, MAX_PAYLOAD_BYTES + 1)
416
+ if obj.unconsumed_tail: # hit the ceiling before the input was exhausted
417
+ raise PayloadTooLargeError(MAX_PAYLOAD_BYTES + 1, MAX_PAYLOAD_BYTES)
418
+ return out + obj.flush()
419
+
420
+
421
+ def read(pdf_bytes: bytes) -> ReadResult:
422
+ """Read + integrity-verify the om.json payload (detection order [OM-XMP-003])."""
423
+ with pikepdf.open(io.BytesIO(pdf_bytes)) as pdf:
424
+ marker = read_marker(pdf)
425
+ stream = _find_ef_stream(pdf)
426
+ if stream is None:
427
+ return ReadResult(present=False, payload=None, hash_valid=None)
428
+ raw = _decoded_payload_bytes(stream)
429
+ payload = parse_hardened(raw) # §J read-side hardening: dup-key + depth guard [Mi18]
430
+ xmp_hash = marker.get("payloadHash") if marker else None
431
+ hash_valid = (hash_bytes(raw) == xmp_hash) if xmp_hash else None
432
+ return ReadResult(
433
+ present=True,
434
+ payload=payload,
435
+ hash_valid=hash_valid,
436
+ source_doc_hash=marker.get("sourceDocHash") if marker else None,
437
+ )