openom-core 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- openom_core/__init__.py +65 -0
- openom_core/canonical.py +166 -0
- openom_core/embed.py +437 -0
- openom_core/errors.py +57 -0
- openom_core/images.py +181 -0
- openom_core/inspect.py +218 -0
- openom_core/om-0.1.schema.json +214 -0
- openom_core/py.typed +0 -0
- openom_core/schema.py +37 -0
- openom_core/summary.py +133 -0
- openom_core/text.py +130 -0
- openom_core/types.py +75 -0
- openom_core/validate.py +475 -0
- openom_core/xmp.py +214 -0
- openom_core-0.1.0.dist-info/METADATA +55 -0
- openom_core-0.1.0.dist-info/RECORD +17 -0
- openom_core-0.1.0.dist-info/WHEEL +4 -0
openom_core/__init__.py
ADDED
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
# SPDX-License-Identifier: MIT
|
|
2
|
+
"""openOM deterministic core.
|
|
3
|
+
|
|
4
|
+
Embed / read / inspect / validate machine-readable, broker-asserted payloads in
|
|
5
|
+
commercial-real-estate offering-memorandum PDFs. Zero inference, ever (CLAUDE.md Rule 1).
|
|
6
|
+
|
|
7
|
+
The stable public surface is re-exported here, so an integrator writes
|
|
8
|
+
``from openom_core import embed, read, validate, load_schema`` (submodule paths keep working too).
|
|
9
|
+
See the spec Part II §A-§E, §H-§J.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from importlib.metadata import PackageNotFoundError
|
|
13
|
+
from importlib.metadata import version as _pkg_version
|
|
14
|
+
|
|
15
|
+
from .canonical import canonicalize, hash_bytes, payload_hash
|
|
16
|
+
from .embed import ReadResult, embed, read, reembed_warnings
|
|
17
|
+
from .errors import CanonicalizationError, Finding, PayloadTooLargeError, Severity
|
|
18
|
+
from .images import ImageManifest, extract_images
|
|
19
|
+
from .inspect import Profile, classify
|
|
20
|
+
from .schema import load_schema
|
|
21
|
+
from .summary import DealSummary, summarize_deal
|
|
22
|
+
from .text import TextResult, extract_text
|
|
23
|
+
from .types import OMPayload, RealEstateListing
|
|
24
|
+
from .validate import Report, Tolerances, validate
|
|
25
|
+
|
|
26
|
+
try: # single source of truth: the installed package metadata (matches `pip show` / the wheel)
|
|
27
|
+
__version__ = _pkg_version("openom-core")
|
|
28
|
+
except PackageNotFoundError: # editable/source tree without an installed dist
|
|
29
|
+
__version__ = "0.0.0+unknown"
|
|
30
|
+
|
|
31
|
+
#: The openOM spec version this library targets. Drift-locked to the schema by tests/test_types.py.
|
|
32
|
+
SPEC_VERSION = "0.1"
|
|
33
|
+
|
|
34
|
+
__all__ = [
|
|
35
|
+
"SPEC_VERSION",
|
|
36
|
+
"__version__",
|
|
37
|
+
# verbs
|
|
38
|
+
"embed",
|
|
39
|
+
"read",
|
|
40
|
+
"reembed_warnings",
|
|
41
|
+
"validate",
|
|
42
|
+
"classify",
|
|
43
|
+
"extract_text",
|
|
44
|
+
"extract_images",
|
|
45
|
+
"canonicalize",
|
|
46
|
+
"hash_bytes",
|
|
47
|
+
"payload_hash",
|
|
48
|
+
"load_schema",
|
|
49
|
+
"summarize_deal",
|
|
50
|
+
# result / option types
|
|
51
|
+
"ReadResult",
|
|
52
|
+
"Report",
|
|
53
|
+
"Tolerances",
|
|
54
|
+
"Profile",
|
|
55
|
+
"TextResult",
|
|
56
|
+
"ImageManifest",
|
|
57
|
+
"DealSummary",
|
|
58
|
+
"Finding",
|
|
59
|
+
"Severity",
|
|
60
|
+
"OMPayload",
|
|
61
|
+
"RealEstateListing",
|
|
62
|
+
# errors
|
|
63
|
+
"CanonicalizationError",
|
|
64
|
+
"PayloadTooLargeError",
|
|
65
|
+
]
|
openom_core/canonical.py
ADDED
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
# SPDX-License-Identifier: MIT
|
|
2
|
+
"""RFC 8785 JSON Canonicalization (JCS) + the openOM integrity hash (spec §C).
|
|
3
|
+
|
|
4
|
+
The keystone of cross-implementation fidelity: two conformant implementations MUST produce
|
|
5
|
+
byte-identical output here, and therefore the same SHA-256. RFC 8785 itself performs no
|
|
6
|
+
Unicode normalization and assumes unique member names (RFC 8785 §3.1), so §C.1 mandates the
|
|
7
|
+
preprocessing done here (NFC, duplicate-key rejection, number-range checks) *before* JCS.
|
|
8
|
+
Serialization (key sorting, minimal escaping, ES number formatting) is delegated to the
|
|
9
|
+
vetted ``rfc8785`` library.
|
|
10
|
+
|
|
11
|
+
Producer vs Consumer contract (§C, §D)
|
|
12
|
+
--------------------------------------
|
|
13
|
+
The **producer** normalizes (NFC), canonicalizes, hashes the resulting bytes, and stores
|
|
14
|
+
*those exact bytes* as the payload plus the hash in the XMP marker. The **consumer** hashes
|
|
15
|
+
the stored bytes *as received* - it does NOT re-canonicalize before verifying. Verification
|
|
16
|
+
is therefore a byte comparison of ``sha256(stored_bytes)`` against the marker hash; it never
|
|
17
|
+
depends on the consumer re-running NFC/JCS. This asymmetry is deliberate: normalization
|
|
18
|
+
happens once, at authoring time, so a consumer on a different platform/library cannot perturb
|
|
19
|
+
the hash. ``canonicalize`` is the producer path; ``hash_bytes`` over stored bytes is the
|
|
20
|
+
consumer path (see :func:`openom_core.embed.read`).
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import copy
|
|
26
|
+
import hashlib
|
|
27
|
+
import json
|
|
28
|
+
import math
|
|
29
|
+
from collections.abc import Mapping
|
|
30
|
+
from typing import Any
|
|
31
|
+
from unicodedata import normalize
|
|
32
|
+
|
|
33
|
+
import rfc8785
|
|
34
|
+
|
|
35
|
+
from .errors import IO_BADUTF8, IO_DUPKEY, IO_NUMRANGE, IO_STRUCTURE, CanonicalizationError
|
|
36
|
+
|
|
37
|
+
#: ECMAScript safe-integer limit; integers beyond this are silently rounded by the number
|
|
38
|
+
#: model, which would be data corruption (§C [OM-CANON-013]).
|
|
39
|
+
MAX_SAFE_INT = 2**53 - 1
|
|
40
|
+
|
|
41
|
+
#: Max nesting depth (§J JSON-hardening guard) - matches the JS parser for cross-impl parity.
|
|
42
|
+
MAX_DEPTH = 64
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _prepare(obj: Any, depth: int = 0) -> Any:
|
|
46
|
+
"""NFC-normalize strings + member names, reject duplicate keys and non-representable numbers.
|
|
47
|
+
|
|
48
|
+
Returns a new structure ready for RFC 8785 serialization. Mutates nothing.
|
|
49
|
+
"""
|
|
50
|
+
if depth > MAX_DEPTH:
|
|
51
|
+
raise CanonicalizationError(IO_STRUCTURE, f"nesting exceeds {MAX_DEPTH}")
|
|
52
|
+
# bool is an int subclass - must be checked first.
|
|
53
|
+
if isinstance(obj, bool):
|
|
54
|
+
return obj
|
|
55
|
+
if isinstance(obj, str):
|
|
56
|
+
# A lone UTF-16 surrogate cannot be encoded as UTF-8; reject it explicitly with a
|
|
57
|
+
# stable code rather than letting the serializer raise a bare UnicodeEncodeError.
|
|
58
|
+
if any(0xD800 <= ord(ch) <= 0xDFFF for ch in obj):
|
|
59
|
+
raise CanonicalizationError(IO_BADUTF8, "string contains an unpaired surrogate")
|
|
60
|
+
return normalize("NFC", obj)
|
|
61
|
+
if isinstance(obj, Mapping):
|
|
62
|
+
out: dict[str, Any] = {}
|
|
63
|
+
for key, value in obj.items():
|
|
64
|
+
if not isinstance(key, str):
|
|
65
|
+
raise CanonicalizationError(IO_DUPKEY, f"non-string member name: {key!r}")
|
|
66
|
+
nkey = normalize("NFC", key)
|
|
67
|
+
if nkey in out:
|
|
68
|
+
raise CanonicalizationError(IO_DUPKEY, f"duplicate member name after NFC: {nkey!r}")
|
|
69
|
+
out[nkey] = _prepare(value, depth + 1)
|
|
70
|
+
return out
|
|
71
|
+
if isinstance(obj, (list, tuple)):
|
|
72
|
+
return [_prepare(item, depth + 1) for item in obj]
|
|
73
|
+
if isinstance(obj, int):
|
|
74
|
+
if abs(obj) > MAX_SAFE_INT:
|
|
75
|
+
raise CanonicalizationError(IO_NUMRANGE, f"integer exceeds 2^53-1: {obj}")
|
|
76
|
+
return obj
|
|
77
|
+
if isinstance(obj, float):
|
|
78
|
+
if not math.isfinite(obj):
|
|
79
|
+
raise CanonicalizationError(IO_NUMRANGE, f"non-finite number: {obj}")
|
|
80
|
+
if obj.is_integer() and abs(obj) > MAX_SAFE_INT:
|
|
81
|
+
raise CanonicalizationError(IO_NUMRANGE, f"float integer exceeds 2^53-1: {obj}")
|
|
82
|
+
return obj
|
|
83
|
+
if obj is None:
|
|
84
|
+
return None
|
|
85
|
+
raise CanonicalizationError(IO_NUMRANGE, f"unsupported JSON type: {type(obj).__name__}")
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _reject_duplicate_pairs(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
|
|
89
|
+
"""json object_pairs_hook: reject duplicate member names (after NFC) rather than last-wins.
|
|
90
|
+
Preserves the ORIGINAL keys (does not normalize) so a reader returns the payload as stored."""
|
|
91
|
+
out: dict[str, Any] = {}
|
|
92
|
+
seen: set[str] = set()
|
|
93
|
+
for key, value in pairs:
|
|
94
|
+
nkey = normalize("NFC", key)
|
|
95
|
+
if nkey in seen:
|
|
96
|
+
raise CanonicalizationError(IO_DUPKEY, f"duplicate member name after NFC: {nkey!r}")
|
|
97
|
+
seen.add(nkey)
|
|
98
|
+
out[key] = value
|
|
99
|
+
return out
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _check_depth(obj: Any, depth: int = 0) -> None:
|
|
103
|
+
if depth > MAX_DEPTH:
|
|
104
|
+
raise CanonicalizationError(IO_STRUCTURE, f"nesting exceeds {MAX_DEPTH}")
|
|
105
|
+
if isinstance(obj, Mapping):
|
|
106
|
+
for v in obj.values():
|
|
107
|
+
_check_depth(v, depth + 1)
|
|
108
|
+
elif isinstance(obj, (list, tuple)):
|
|
109
|
+
for v in obj:
|
|
110
|
+
_check_depth(v, depth + 1)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def parse_hardened(raw: bytes | str) -> Any:
|
|
114
|
+
"""Parse an om.json payload with the §J read-side hardening the write path enforces
|
|
115
|
+
([OM-CANON-009/010]): reject duplicate member names and over-deep nesting. Used by the reusable
|
|
116
|
+
``read`` verb so a self-hoster calling core on untrusted PDFs gets the MCP invariants."""
|
|
117
|
+
obj = json.loads(raw, object_pairs_hook=_reject_duplicate_pairs)
|
|
118
|
+
_check_depth(obj)
|
|
119
|
+
return obj
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def canonicalize(payload: Mapping[str, Any]) -> bytes:
|
|
123
|
+
"""Serialize a payload to its RFC 8785 JCS bytes (UTF-8, no BOM). Producer path.
|
|
124
|
+
|
|
125
|
+
Applies the §C.1 preprocessing (NFC, unique keys, number range) then delegates
|
|
126
|
+
serialization to ``rfc8785``. The top level MUST be a JSON object (§C.10); a bare
|
|
127
|
+
array/scalar is rejected rather than hashed.
|
|
128
|
+
"""
|
|
129
|
+
if not isinstance(payload, Mapping):
|
|
130
|
+
raise CanonicalizationError(
|
|
131
|
+
IO_STRUCTURE, f"top-level value must be an object, got {type(payload).__name__}"
|
|
132
|
+
)
|
|
133
|
+
prepared = _prepare(payload)
|
|
134
|
+
try:
|
|
135
|
+
return rfc8785.dumps(prepared)
|
|
136
|
+
except CanonicalizationError:
|
|
137
|
+
raise
|
|
138
|
+
except Exception as exc: # noqa: BLE001 - normalize any serializer failure to our code
|
|
139
|
+
raise CanonicalizationError(IO_NUMRANGE, str(exc)) from exc
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def hash_bytes(data: bytes) -> str:
|
|
143
|
+
"""The openOM integrity hash of already-canonical bytes: ``sha256:<lowercase-hex>``.
|
|
144
|
+
|
|
145
|
+
Used on both the write path (over the bytes just produced) and the read/verify path
|
|
146
|
+
(over the decompressed stored bytes, as received - no re-canonicalization; §C, §D).
|
|
147
|
+
"""
|
|
148
|
+
return "sha256:" + hashlib.sha256(data).hexdigest()
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def strip_signature(payload: Mapping[str, Any]) -> dict[str, Any]:
|
|
152
|
+
"""Return a deep copy with ``meta.signature`` *removed* (not nulled), per [OM-CANON-003].
|
|
153
|
+
|
|
154
|
+
In 0.1 the signature is always absent/null, so this is a no-op on real payloads; it exists
|
|
155
|
+
so that adding a signature in a future version does not change the integrity hash.
|
|
156
|
+
"""
|
|
157
|
+
out = copy.deepcopy(dict(payload))
|
|
158
|
+
meta = out.get("meta")
|
|
159
|
+
if isinstance(meta, dict) and "signature" in meta:
|
|
160
|
+
meta.pop("signature", None)
|
|
161
|
+
return out
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def payload_hash(payload: Mapping[str, Any]) -> str:
|
|
165
|
+
"""Convenience: the integrity hash of a payload object (strip signature → JCS → sha256)."""
|
|
166
|
+
return hash_bytes(canonicalize(strip_signature(payload)))
|
openom_core/embed.py
ADDED
|
@@ -0,0 +1,437 @@
|
|
|
1
|
+
# SPDX-License-Identifier: MIT
|
|
2
|
+
"""Embed / read the om.json payload in a PDF via pikepdf (spec §D).
|
|
3
|
+
|
|
4
|
+
Non-destructive by construction: pikepdf appends the embedded file + XMP marker without
|
|
5
|
+
touching page content. The catalog ``/AF`` array is added manually - assigning to
|
|
6
|
+
``Pdf.attachments`` populates the ``/EmbeddedFiles`` name tree but NOT ``/AF`` ([OM-EMB-002]).
|
|
7
|
+
The exact JCS bytes are stored verbatim ([OM-EMB-010]); the integrity hash is over those bytes.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import io
|
|
13
|
+
import os
|
|
14
|
+
import re
|
|
15
|
+
import tempfile
|
|
16
|
+
import zlib
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
import pikepdf
|
|
21
|
+
|
|
22
|
+
from .canonical import canonicalize, hash_bytes, parse_hardened, strip_signature
|
|
23
|
+
from .errors import Finding, PayloadTooLargeError, SignedEmbedError
|
|
24
|
+
from .xmp import _marker_props, read_marker, render_marker_xml, write_marker
|
|
25
|
+
|
|
26
|
+
#: Decompressed-payload cap (§J [OM-SEC-002]).
|
|
27
|
+
MAX_PAYLOAD_BYTES = 5_000_000
|
|
28
|
+
|
|
29
|
+
PAYLOAD_NAME = "om.json"
|
|
30
|
+
MIME = "application/ld+json"
|
|
31
|
+
SUBTYPE = pikepdf.Name("/application/ld+json") # serialized name-escaped as /application#2Fld+json
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class ReadResult:
|
|
36
|
+
"""Result of reading a payload from a PDF (§I om_read shape)."""
|
|
37
|
+
|
|
38
|
+
present: bool
|
|
39
|
+
payload: dict[str, Any] | None
|
|
40
|
+
hash_valid: bool | None
|
|
41
|
+
origin_verified: None = None # read-time origin check is a Consumer concern; null in core
|
|
42
|
+
signature_valid: None = None # reserved (§10 layer 4)
|
|
43
|
+
source_doc_hash: str | None = None # #5: marker sourceDocHash (provenance of the source PDF)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _remove_existing(pdf: pikepdf.Pdf) -> None:
|
|
47
|
+
"""Remove any existing om.json attachment and its /AF reference (idempotent embed)."""
|
|
48
|
+
if PAYLOAD_NAME not in pdf.attachments:
|
|
49
|
+
return
|
|
50
|
+
old_objgen = pdf.attachments[PAYLOAD_NAME].obj.objgen
|
|
51
|
+
# Filter /AF while the referenced object is still live, then delete the attachment.
|
|
52
|
+
if "/AF" in pdf.Root:
|
|
53
|
+
kept = [f for f in pdf.Root.AF if f.objgen != old_objgen]
|
|
54
|
+
if kept:
|
|
55
|
+
pdf.Root.AF = pikepdf.Array(kept)
|
|
56
|
+
else:
|
|
57
|
+
del pdf.Root.AF
|
|
58
|
+
del pdf.attachments[PAYLOAD_NAME]
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _ensure_af(pdf: pikepdf.Pdf, spec: pikepdf.Object) -> None:
|
|
62
|
+
"""Ensure the catalog /AF array references the payload's Filespec ([OM-EMB-002])."""
|
|
63
|
+
if "/AF" not in pdf.Root:
|
|
64
|
+
pdf.Root.AF = pikepdf.Array([spec])
|
|
65
|
+
return
|
|
66
|
+
if all(f.objgen != spec.objgen for f in pdf.Root.AF):
|
|
67
|
+
pdf.Root.AF.append(spec)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _set_subtype(spec: pikepdf.Object) -> None:
|
|
71
|
+
"""Set /Subtype application/ld+json on the embedded-file stream(s) ([OM-EMB-004])."""
|
|
72
|
+
ef = spec.EF
|
|
73
|
+
for key in ("/F", "/UF"):
|
|
74
|
+
if key in ef:
|
|
75
|
+
ef[key].Subtype = SUBTYPE
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def input_encrypted(pdf_bytes: bytes) -> bool:
|
|
79
|
+
"""True if the input PDF is encrypted (permission encryption with an empty user password, which
|
|
80
|
+
pikepdf opens transparently and embed() then writes out UNENCRYPTED - a silent security-posture
|
|
81
|
+
change worth signaling to the author). A password-protected PDF raises on open and never reaches
|
|
82
|
+
here, so this is specifically the restrictions-only case (#4)."""
|
|
83
|
+
try:
|
|
84
|
+
with pikepdf.open(io.BytesIO(pdf_bytes)) as pdf:
|
|
85
|
+
return bool(pdf.is_encrypted)
|
|
86
|
+
except pikepdf.PasswordError:
|
|
87
|
+
return False
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _is_signed(pdf: pikepdf.Pdf) -> bool:
|
|
91
|
+
"""True if the PDF carries a digital signature (§10 layer 4 / #3 [OM-EMB-020]).
|
|
92
|
+
|
|
93
|
+
A full-rewrite save invalidates a byte-range signature; when this is True, embed uses an
|
|
94
|
+
incremental-update save instead, appending the payload so the signed bytes stay untouched.
|
|
95
|
+
Detects the AcroForm ``SigFlags`` "signatures exist" bit, any ``/FT /Sig`` field carrying a
|
|
96
|
+
``/V``, or a ``/Perms`` (DocMDP/UR) entry - covering approval and certification signatures.
|
|
97
|
+
"""
|
|
98
|
+
root = pdf.Root
|
|
99
|
+
if "/Perms" in root:
|
|
100
|
+
return True
|
|
101
|
+
acro = root.get("/AcroForm")
|
|
102
|
+
if acro is None:
|
|
103
|
+
return False
|
|
104
|
+
sig_flags = acro.get("/SigFlags")
|
|
105
|
+
if sig_flags is not None and int(sig_flags) & 1:
|
|
106
|
+
return True
|
|
107
|
+
fields = acro.get("/Fields")
|
|
108
|
+
if fields is None:
|
|
109
|
+
return False
|
|
110
|
+
return any(f.get("/FT") == pikepdf.Name.Sig and "/V" in f for f in fields)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
@dataclass
|
|
114
|
+
class _EmbedFields:
|
|
115
|
+
data: bytes
|
|
116
|
+
payload_hash: str
|
|
117
|
+
spec_version: str
|
|
118
|
+
supersedes: str | None
|
|
119
|
+
source_doc_hash: str
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _plan_embed(pdf: pikepdf.Pdf, pdf_bytes: bytes, payload: dict[str, Any]) -> _EmbedFields:
|
|
123
|
+
"""Compute the payload bytes + marker fields (supersedes/sourceDocHash) - shared by both the
|
|
124
|
+
full-rewrite and incremental save paths so a signed OM gets identical provenance semantics."""
|
|
125
|
+
# signature excluded from the integrity preimage ([OM-CANON-003])
|
|
126
|
+
data = canonicalize(strip_signature(payload))
|
|
127
|
+
if len(data) > MAX_PAYLOAD_BYTES:
|
|
128
|
+
raise PayloadTooLargeError(len(data), MAX_PAYLOAD_BYTES)
|
|
129
|
+
payload_hash = hash_bytes(data)
|
|
130
|
+
# §D.4 re-embed semantics:
|
|
131
|
+
# - a *different* prior payload is superseded (supersedes = prior hash);
|
|
132
|
+
# - an *identical* re-embed carries the prior supersedes forward (no lineage wipe, no
|
|
133
|
+
# self-supersede); a first embed has no predecessor (supersedes = None).
|
|
134
|
+
prior = read_marker(pdf)
|
|
135
|
+
prior_hash = prior.get("payloadHash") if prior else None
|
|
136
|
+
prior_supersedes = prior.get("supersedes") if prior else None
|
|
137
|
+
if prior_hash and prior_hash != payload_hash:
|
|
138
|
+
supersedes: str | None = prior_hash
|
|
139
|
+
elif prior_hash == payload_hash:
|
|
140
|
+
supersedes = prior_supersedes
|
|
141
|
+
else:
|
|
142
|
+
supersedes = None
|
|
143
|
+
# #5: sourceDocHash identifies the underlying source PDF, held STABLE across reprices - computed
|
|
144
|
+
# once (first embed) and carried forward from the prior marker on every re-embed.
|
|
145
|
+
prior_source = prior.get("sourceDocHash") if prior else None
|
|
146
|
+
source_doc_hash = prior_source or hash_bytes(pdf_bytes)
|
|
147
|
+
return _EmbedFields(
|
|
148
|
+
data=data,
|
|
149
|
+
payload_hash=payload_hash,
|
|
150
|
+
spec_version=str(payload.get("specVersion", "0.1")),
|
|
151
|
+
supersedes=supersedes,
|
|
152
|
+
source_doc_hash=source_doc_hash,
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def embed(
|
|
157
|
+
pdf_bytes: bytes, payload: dict[str, Any], *, asserted_date: str, badge: bool = False
|
|
158
|
+
) -> bytes:
|
|
159
|
+
"""Embed ``payload`` as om.json and return the new PDF bytes. Never mutates the input.
|
|
160
|
+
|
|
161
|
+
A *signed* input (#3 [OM-EMB-020]) is embedded via an incremental-update save - the payload is
|
|
162
|
+
appended after the signed byte range so the signature stays cryptographically intact - rather
|
|
163
|
+
than the default full-rewrite (which would invalidate it). Both paths write the identical
|
|
164
|
+
payload bytes and XMP marker, so the result reads the same regardless of the save method.
|
|
165
|
+
"""
|
|
166
|
+
with pikepdf.open(io.BytesIO(pdf_bytes)) as pdf:
|
|
167
|
+
fields = _plan_embed(pdf, pdf_bytes, payload)
|
|
168
|
+
signed = _is_signed(pdf)
|
|
169
|
+
|
|
170
|
+
if signed:
|
|
171
|
+
return _embed_incremental(pdf_bytes, fields, asserted_date)
|
|
172
|
+
|
|
173
|
+
with pikepdf.open(io.BytesIO(pdf_bytes)) as pdf:
|
|
174
|
+
_remove_existing(pdf)
|
|
175
|
+
# pikepdf's stub marks description/filename/dates as required; runtime defaults them.
|
|
176
|
+
# We intentionally omit dates for determinism (§D [OM-EMB-011]).
|
|
177
|
+
filespec = pikepdf.AttachedFileSpec(pdf, fields.data, mime_type=MIME) # type: ignore[call-arg]
|
|
178
|
+
filespec.relationship = pikepdf.Name.Data # /AFRelationship (kwarg missing from stub)
|
|
179
|
+
pdf.attachments[PAYLOAD_NAME] = filespec
|
|
180
|
+
spec_obj = pdf.attachments[PAYLOAD_NAME].obj
|
|
181
|
+
_ensure_af(pdf, spec_obj)
|
|
182
|
+
_set_subtype(spec_obj)
|
|
183
|
+
write_marker(
|
|
184
|
+
pdf,
|
|
185
|
+
spec_version=fields.spec_version,
|
|
186
|
+
payload_filename=PAYLOAD_NAME,
|
|
187
|
+
payload_hash=fields.payload_hash,
|
|
188
|
+
asserted_date=asserted_date,
|
|
189
|
+
supersedes=fields.supersedes,
|
|
190
|
+
source_doc_hash=fields.source_doc_hash,
|
|
191
|
+
)
|
|
192
|
+
out = io.BytesIO()
|
|
193
|
+
pdf.save(out, deterministic_id=True)
|
|
194
|
+
return out.getvalue()
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _embed_incremental(pdf_bytes: bytes, fields: _EmbedFields, asserted_date: str) -> bytes:
|
|
198
|
+
"""Append the payload to a *signed* PDF via a fitz incremental-update save (#3 [OM-EMB-020]).
|
|
199
|
+
|
|
200
|
+
Builds the om.json embedded-file stream, an indirect /Filespec (/AFRelationship /Data,
|
|
201
|
+
/Subtype application/ld+json), the /EmbeddedFiles name-tree entry, the catalog /AF reference,
|
|
202
|
+
and the XMP marker - then saves incrementally so the original signed bytes are preserved as a
|
|
203
|
+
byte-exact prefix. The marker bytes come from the same ``render_marker_xml`` the full-rewrite
|
|
204
|
+
path uses, so the two producers agree. fitz's incremental save requires a real file whose bytes
|
|
205
|
+
equal the source, so the work happens in a temp file.
|
|
206
|
+
"""
|
|
207
|
+
# PyMuPDF is the optional [render] extra; the signed-OM incremental path is the one embed case
|
|
208
|
+
# that needs it. Imported lazily with a clear hint so the pikepdf-only core stays MIT-clean.
|
|
209
|
+
try:
|
|
210
|
+
import pymupdf
|
|
211
|
+
except ImportError as exc:
|
|
212
|
+
raise SignedEmbedError(
|
|
213
|
+
"embedding a signed PDF needs PyMuPDF: pip install 'openom-core[render]'"
|
|
214
|
+
) from exc
|
|
215
|
+
|
|
216
|
+
props = _marker_props(
|
|
217
|
+
spec_version=fields.spec_version,
|
|
218
|
+
payload_filename=PAYLOAD_NAME,
|
|
219
|
+
payload_hash=fields.payload_hash,
|
|
220
|
+
asserted_date=asserted_date,
|
|
221
|
+
supersedes=fields.supersedes,
|
|
222
|
+
source_doc_hash=fields.source_doc_hash,
|
|
223
|
+
)
|
|
224
|
+
# A TemporaryDirectory (vs NamedTemporaryFile + os.unlink) so cleanup can't mask the real error
|
|
225
|
+
# with a Windows "file in use" (WinError 32) in the finally block.
|
|
226
|
+
with tempfile.TemporaryDirectory(prefix="openom_signed_") as tmpdir:
|
|
227
|
+
path = os.path.join(tmpdir, "in.pdf")
|
|
228
|
+
with open(path, "wb") as fh:
|
|
229
|
+
fh.write(pdf_bytes)
|
|
230
|
+
doc = pymupdf.open(path)
|
|
231
|
+
try:
|
|
232
|
+
# If pymupdf had to REBUILD the xref to open this file, an incremental append is not a
|
|
233
|
+
# byte-exact extension of the original - the signed prefix would change and the
|
|
234
|
+
# signature break. Refuse cleanly (OM-EMB-021) rather than ship an invalid signature.
|
|
235
|
+
if getattr(doc, "is_repaired", False):
|
|
236
|
+
raise SignedEmbedError(
|
|
237
|
+
"this signed PDF needed its cross-reference table rebuilt to open, so an "
|
|
238
|
+
"incremental (signature-preserving) embed is not safe; embed an unsigned copy "
|
|
239
|
+
"or re-issue the signature after embedding"
|
|
240
|
+
)
|
|
241
|
+
cat = doc.pdf_catalog()
|
|
242
|
+
# 1. embedded-file stream (verbatim JCS bytes; pikepdf read decodes the filter)
|
|
243
|
+
ef = doc.get_new_xref()
|
|
244
|
+
doc.update_object(ef, "<< /Type /EmbeddedFile /Subtype /application#2Fld+json >>")
|
|
245
|
+
doc.update_stream(ef, fields.data, compress=True)
|
|
246
|
+
# 2. indirect /Filespec ([OM-EMB-002]/[OM-EMB-004])
|
|
247
|
+
spec = doc.get_new_xref()
|
|
248
|
+
doc.update_object(
|
|
249
|
+
spec,
|
|
250
|
+
f"<< /Type /Filespec /F ({PAYLOAD_NAME}) /UF ({PAYLOAD_NAME}) "
|
|
251
|
+
f"/AFRelationship /Data /Desc (openOM payload) "
|
|
252
|
+
f"/EF << /F {ef} 0 R /UF {ef} 0 R >> >>",
|
|
253
|
+
)
|
|
254
|
+
# 3. /EmbeddedFiles name tree - drop any prior om.json, keep other attachments
|
|
255
|
+
entries = [
|
|
256
|
+
f"({nm}) {xr} 0 R"
|
|
257
|
+
for nm, xr in _name_tree_entries(doc, cat)
|
|
258
|
+
if nm != PAYLOAD_NAME
|
|
259
|
+
]
|
|
260
|
+
entries.append(f"({PAYLOAD_NAME}) {spec} 0 R")
|
|
261
|
+
doc.xref_set_key(cat, "Names/EmbeddedFiles/Names", "[" + " ".join(entries) + "]")
|
|
262
|
+
# 4. catalog /AF - drop any prior om.json filespec ref (by resolving each ref's /F,
|
|
263
|
+
# since a re-embed's prior filespec has a different xref), append ours
|
|
264
|
+
af = [r for r in _af_refs(doc, cat) if not _is_om_filespec(doc, r)]
|
|
265
|
+
af.append(f"{spec} 0 R")
|
|
266
|
+
doc.xref_set_key(cat, "AF", "[" + " ".join(af) + "]")
|
|
267
|
+
# 5. XMP marker (identical bytes to the full-rewrite path)
|
|
268
|
+
existing = _existing_metadata_xml(doc, cat)
|
|
269
|
+
meta = doc.get_new_xref()
|
|
270
|
+
doc.update_object(meta, "<< /Type /Metadata /Subtype /XML >>")
|
|
271
|
+
marker_xml = render_marker_xml(existing, props).encode("utf-8")
|
|
272
|
+
doc.update_stream(meta, marker_xml, compress=False)
|
|
273
|
+
doc.xref_set_key(cat, "Metadata", f"{meta} 0 R")
|
|
274
|
+
try:
|
|
275
|
+
doc.saveIncr()
|
|
276
|
+
except Exception as exc: # noqa: BLE001 - any fitz save failure -> typed refusal
|
|
277
|
+
raise SignedEmbedError(
|
|
278
|
+
f"incremental save failed on this signed PDF: {exc}"
|
|
279
|
+
) from exc
|
|
280
|
+
finally:
|
|
281
|
+
doc.close()
|
|
282
|
+
with open(path, "rb") as fh:
|
|
283
|
+
out = fh.read()
|
|
284
|
+
# Postcondition the whole feature promises: the original signed bytes stay a byte-exact prefix,
|
|
285
|
+
# so every byte the /ByteRange signs is unchanged. Never return a doc that broke it.
|
|
286
|
+
if not out.startswith(pdf_bytes):
|
|
287
|
+
raise SignedEmbedError("incremental embed did not preserve the signed byte prefix")
|
|
288
|
+
return out
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def _name_tree_entries(doc: Any, cat: int) -> list[tuple[str, str]]:
|
|
292
|
+
"""(name, 'N 0 R') pairs already in /Names/EmbeddedFiles/Names, or [] if none."""
|
|
293
|
+
kind, val = doc.xref_get_key(cat, "Names/EmbeddedFiles/Names")
|
|
294
|
+
if kind != "array":
|
|
295
|
+
return []
|
|
296
|
+
return re.findall(r"\(([^)]*)\)\s*(\d+ 0 R)", val)
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def _af_refs(doc: Any, cat: int) -> list[str]:
|
|
300
|
+
"""Existing catalog /AF indirect refs ('N 0 R'), or [] if none."""
|
|
301
|
+
kind, val = doc.xref_get_key(cat, "AF")
|
|
302
|
+
if kind != "array":
|
|
303
|
+
return []
|
|
304
|
+
return re.findall(r"\d+ 0 R", val)
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def _is_om_filespec(doc: Any, ref: str) -> bool:
|
|
308
|
+
"""True if the /AF ref 'N 0 R' points to a filespec whose /F is our om.json (drop it on
|
|
309
|
+
re-embed so /AF never stacks duplicate payload references)."""
|
|
310
|
+
m = re.match(r"(\d+) 0 R", ref)
|
|
311
|
+
if not m:
|
|
312
|
+
return False
|
|
313
|
+
kind, val = doc.xref_get_key(int(m.group(1)), "F")
|
|
314
|
+
return bool(kind == "string" and val == PAYLOAD_NAME)
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def _existing_metadata_xml(doc: Any, cat: int) -> str | None:
|
|
318
|
+
"""The current catalog /Metadata XML, or None - so our marker merges alongside existing XMP."""
|
|
319
|
+
kind, val = doc.xref_get_key(cat, "Metadata")
|
|
320
|
+
if kind != "xref":
|
|
321
|
+
return None
|
|
322
|
+
meta_xref = int(re.match(r"(\d+) 0 R", val).group(1)) # type: ignore[union-attr]
|
|
323
|
+
raw = doc.xref_stream(meta_xref)
|
|
324
|
+
return raw.decode("utf-8", "replace") if raw else None
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
def reembed_warnings(
|
|
328
|
+
pdf_bytes: bytes, payload: dict[str, Any], *, asserted_date: str
|
|
329
|
+
) -> list[Finding]:
|
|
330
|
+
"""Non-blocking re-embed warnings against an existing PDF (§H). Pure; ``embed`` stays
|
|
331
|
+
bytes→bytes, so callers (CLI/MCP) compose this to surface provenance issues.
|
|
332
|
+
|
|
333
|
+
OMW-W051: the new assertedDate precedes the payload it would supersede (time going
|
|
334
|
+
backwards on a reprice). Warnings never block.
|
|
335
|
+
"""
|
|
336
|
+
with pikepdf.open(io.BytesIO(pdf_bytes)) as pdf:
|
|
337
|
+
prior = read_marker(pdf)
|
|
338
|
+
if not prior:
|
|
339
|
+
return []
|
|
340
|
+
payload_hash = hash_bytes(canonicalize(strip_signature(payload)))
|
|
341
|
+
prior_hash = prior.get("payloadHash")
|
|
342
|
+
prior_date = prior.get("assertedDate")
|
|
343
|
+
out: list[Finding] = []
|
|
344
|
+
if prior_hash and prior_hash != payload_hash and prior_date and asserted_date < prior_date:
|
|
345
|
+
out.append(
|
|
346
|
+
Finding(
|
|
347
|
+
"OMW-W051",
|
|
348
|
+
"warning",
|
|
349
|
+
"/assertedDate",
|
|
350
|
+
"assertedDate precedes the superseded payload's assertedDate",
|
|
351
|
+
expected=prior_date,
|
|
352
|
+
actual=asserted_date,
|
|
353
|
+
)
|
|
354
|
+
)
|
|
355
|
+
return out
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def _find_ef_stream(pdf: pikepdf.Pdf) -> pikepdf.Object | None:
|
|
359
|
+
"""Locate the om.json embedded-file stream in spec detection order ([OM-XMP-003]).
|
|
360
|
+
|
|
361
|
+
Authoritative path: catalog ``/AF`` → Filespec (``/UF`` or ``/F`` == om.json) → ``/EF``.
|
|
362
|
+
Falls back to the ``/EmbeddedFiles`` name tree for producers that populate only that.
|
|
363
|
+
"""
|
|
364
|
+
|
|
365
|
+
def _ef_of(spec: pikepdf.Object) -> pikepdf.Object | None:
|
|
366
|
+
if "/EF" not in spec:
|
|
367
|
+
return None
|
|
368
|
+
ef = spec.EF
|
|
369
|
+
for key in ("/UF", "/F"):
|
|
370
|
+
if key in ef:
|
|
371
|
+
return ef[key]
|
|
372
|
+
return None
|
|
373
|
+
|
|
374
|
+
if "/AF" in pdf.Root:
|
|
375
|
+
for spec in pdf.Root.AF:
|
|
376
|
+
names = {str(spec[k]) for k in ("/UF", "/F") if k in spec}
|
|
377
|
+
if PAYLOAD_NAME in names:
|
|
378
|
+
stream = _ef_of(spec)
|
|
379
|
+
if stream is not None:
|
|
380
|
+
return stream
|
|
381
|
+
if PAYLOAD_NAME in pdf.attachments: # fallback: EmbeddedFiles name tree
|
|
382
|
+
return _ef_of(pdf.attachments[PAYLOAD_NAME].obj)
|
|
383
|
+
return None
|
|
384
|
+
|
|
385
|
+
|
|
386
|
+
def _decoded_payload_bytes(stream: pikepdf.Object) -> bytes:
|
|
387
|
+
"""Decode the EF stream to raw payload bytes, bounding size *before* full materialization.
|
|
388
|
+
|
|
389
|
+
A malicious PDF can hide a decompression bomb in a tiny compressed stream. We read the
|
|
390
|
+
stored (compressed) bytes - always bounded by the file itself - reject if already over the
|
|
391
|
+
cap, then inflate with a hard ceiling rather than decompressing unbounded into memory.
|
|
392
|
+
"""
|
|
393
|
+
raw = bytes(stream.read_raw_bytes())
|
|
394
|
+
if len(raw) > MAX_PAYLOAD_BYTES:
|
|
395
|
+
raise PayloadTooLargeError(len(raw), MAX_PAYLOAD_BYTES)
|
|
396
|
+
|
|
397
|
+
filt = stream.get("/Filter")
|
|
398
|
+
names = [str(filt)] if isinstance(filt, pikepdf.Name) else [str(n) for n in (filt or [])]
|
|
399
|
+
# Plain FlateDecode (no predictor) - the common case for our writes and pdf-lib - can be
|
|
400
|
+
# inflated with a bounded zlib object. Anything exotic (predictors, other filters) defers
|
|
401
|
+
# to pikepdf's decoder, still guarded by the compressed-size cap above and a post-check.
|
|
402
|
+
if names == ["/FlateDecode"] and "/DecodeParms" not in stream and "/DP" not in stream:
|
|
403
|
+
data = _bounded_inflate(raw)
|
|
404
|
+
elif not names:
|
|
405
|
+
data = raw # stored uncompressed
|
|
406
|
+
else:
|
|
407
|
+
data = bytes(stream.read_bytes())
|
|
408
|
+
if len(data) > MAX_PAYLOAD_BYTES:
|
|
409
|
+
raise PayloadTooLargeError(len(data), MAX_PAYLOAD_BYTES)
|
|
410
|
+
return data
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def _bounded_inflate(raw: bytes) -> bytes:
|
|
414
|
+
obj = zlib.decompressobj()
|
|
415
|
+
out = obj.decompress(raw, MAX_PAYLOAD_BYTES + 1)
|
|
416
|
+
if obj.unconsumed_tail: # hit the ceiling before the input was exhausted
|
|
417
|
+
raise PayloadTooLargeError(MAX_PAYLOAD_BYTES + 1, MAX_PAYLOAD_BYTES)
|
|
418
|
+
return out + obj.flush()
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
def read(pdf_bytes: bytes) -> ReadResult:
|
|
422
|
+
"""Read + integrity-verify the om.json payload (detection order [OM-XMP-003])."""
|
|
423
|
+
with pikepdf.open(io.BytesIO(pdf_bytes)) as pdf:
|
|
424
|
+
marker = read_marker(pdf)
|
|
425
|
+
stream = _find_ef_stream(pdf)
|
|
426
|
+
if stream is None:
|
|
427
|
+
return ReadResult(present=False, payload=None, hash_valid=None)
|
|
428
|
+
raw = _decoded_payload_bytes(stream)
|
|
429
|
+
payload = parse_hardened(raw) # §J read-side hardening: dup-key + depth guard [Mi18]
|
|
430
|
+
xmp_hash = marker.get("payloadHash") if marker else None
|
|
431
|
+
hash_valid = (hash_bytes(raw) == xmp_hash) if xmp_hash else None
|
|
432
|
+
return ReadResult(
|
|
433
|
+
present=True,
|
|
434
|
+
payload=payload,
|
|
435
|
+
hash_valid=hash_valid,
|
|
436
|
+
source_doc_hash=marker.get("sourceDocHash") if marker else None,
|
|
437
|
+
)
|