groundgate 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- groundgate/__init__.py +22 -0
- groundgate/adapters/__init__.py +1 -0
- groundgate/adapters/langextract.py +123 -0
- groundgate/admit.py +333 -0
- groundgate/canonical.py +120 -0
- groundgate/cli.py +184 -0
- groundgate/codes.py +36 -0
- groundgate/extract/__init__.py +74 -0
- groundgate/extract/layout.py +128 -0
- groundgate/extract/markup.py +180 -0
- groundgate/extract/pdf.py +210 -0
- groundgate/model.py +213 -0
- groundgate/py.typed +0 -0
- groundgate/report.py +349 -0
- groundgate/text.py +207 -0
- groundgate-0.1.0.dist-info/METADATA +275 -0
- groundgate-0.1.0.dist-info/RECORD +20 -0
- groundgate-0.1.0.dist-info/WHEEL +4 -0
- groundgate-0.1.0.dist-info/entry_points.txt +2 -0
- groundgate-0.1.0.dist-info/licenses/LICENSE +202 -0
groundgate/__init__.py
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""Deterministic admission for LLM-extracted facts."""
|
|
2
|
+
|
|
3
|
+
from .admit import Verification, admit, verify
|
|
4
|
+
from .canonical import digest, jcs
|
|
5
|
+
from .model import Decision, Field, PacketError, Policy, Receipt, Schema
|
|
6
|
+
|
|
7
|
+
__version__ = "0.1.0"
|
|
8
|
+
|
|
9
|
+
__all__ = [
|
|
10
|
+
"Decision",
|
|
11
|
+
"Field",
|
|
12
|
+
"PacketError",
|
|
13
|
+
"Policy",
|
|
14
|
+
"Receipt",
|
|
15
|
+
"Schema",
|
|
16
|
+
"Verification",
|
|
17
|
+
"__version__",
|
|
18
|
+
"admit",
|
|
19
|
+
"digest",
|
|
20
|
+
"jcs",
|
|
21
|
+
"verify",
|
|
22
|
+
]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Adapters from other extraction tools' output to groundgate candidates."""
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""Admit LangExtract results.
|
|
2
|
+
|
|
3
|
+
LangExtract checks that each ``extraction_text`` is found in the source and records where
|
|
4
|
+
(``char_interval``). It does not check the typed value you asked for in the attributes. This
|
|
5
|
+
adapter turns each extraction into a groundgate candidate so the value and unit are checked at
|
|
6
|
+
that location::
|
|
7
|
+
|
|
8
|
+
result = lx.extract(text_or_documents=text, prompt_description=..., examples=...)
|
|
9
|
+
receipt = admit_document(result, schema)
|
|
10
|
+
|
|
11
|
+
Works on ``lx.data.AnnotatedDocument`` objects and on the dicts LangExtract writes to JSONL
|
|
12
|
+
(``lx.io.save_annotated_documents``). LangExtract itself is not imported.
|
|
13
|
+
|
|
14
|
+
Mapping, per extraction:
|
|
15
|
+
|
|
16
|
+
- ``field``: ``extraction_class``, renamed through ``fields`` when given;
|
|
17
|
+
- ``value``: the ``value`` attribute, or ``extraction_text`` when there is none;
|
|
18
|
+
- ``unit``: the ``unit`` attribute, when present;
|
|
19
|
+
- ``evidence``: ``char_interval`` converted to UTF-8 byte offsets, quoting ``extraction_text``.
|
|
20
|
+
An extraction LangExtract could not align has no evidence and is rejected ``NO_EVIDENCE``.
|
|
21
|
+
|
|
22
|
+
``alignment_status`` and ``extraction_class`` are kept on the candidate for the report; the
|
|
23
|
+
checks ignore them.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
from collections.abc import Mapping
|
|
29
|
+
from typing import Any
|
|
30
|
+
|
|
31
|
+
from ..admit import admit
|
|
32
|
+
from ..canonical import Offsets, is_nfc
|
|
33
|
+
from ..model import PacketError, Policy, Receipt, Schema
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _get(obj: Any, name: str) -> Any:
|
|
37
|
+
if isinstance(obj, Mapping):
|
|
38
|
+
return obj.get(name)
|
|
39
|
+
return getattr(obj, name, None)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _status(raw: Any) -> str | None:
|
|
43
|
+
if raw is None:
|
|
44
|
+
return None
|
|
45
|
+
return str(getattr(raw, "value", raw)) # an AlignmentStatus enum or its string value
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def document_text(document: Any) -> str:
|
|
49
|
+
text = _get(document, "text")
|
|
50
|
+
if not isinstance(text, str):
|
|
51
|
+
raise PacketError("the LangExtract document has no text")
|
|
52
|
+
if not is_nfc(text):
|
|
53
|
+
raise PacketError(
|
|
54
|
+
"the LangExtract document text is not NFC; normalise it before calling lx.extract "
|
|
55
|
+
"(unicodedata.normalize('NFC', text)) so offsets line up"
|
|
56
|
+
)
|
|
57
|
+
return text
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def to_candidates(
|
|
61
|
+
document: Any,
|
|
62
|
+
*,
|
|
63
|
+
fields: Mapping[str, str] | None = None,
|
|
64
|
+
value_attribute: str = "value",
|
|
65
|
+
unit_attribute: str = "unit",
|
|
66
|
+
) -> list[dict[str, Any]]:
|
|
67
|
+
"""One groundgate candidate per extraction in a LangExtract ``AnnotatedDocument``."""
|
|
68
|
+
text = document_text(document)
|
|
69
|
+
offsets = Offsets(text)
|
|
70
|
+
out = []
|
|
71
|
+
for i, x in enumerate(_get(document, "extractions") or []):
|
|
72
|
+
cls = _get(x, "extraction_class")
|
|
73
|
+
quote = _get(x, "extraction_text")
|
|
74
|
+
attrs = _get(x, "attributes") or {}
|
|
75
|
+
cand: dict[str, Any] = {
|
|
76
|
+
"id": i,
|
|
77
|
+
"field": (fields or {}).get(cls, cls),
|
|
78
|
+
"value": attrs.get(value_attribute, quote),
|
|
79
|
+
"extraction_class": cls,
|
|
80
|
+
"alignment_status": _status(_get(x, "alignment_status")),
|
|
81
|
+
}
|
|
82
|
+
if attrs.get(unit_attribute) is not None:
|
|
83
|
+
cand["unit"] = attrs[unit_attribute]
|
|
84
|
+
interval = _get(x, "char_interval")
|
|
85
|
+
start, end = _get(interval, "start_pos"), _get(interval, "end_pos")
|
|
86
|
+
if isinstance(start, int) and isinstance(end, int):
|
|
87
|
+
if not 0 <= start <= end <= len(text):
|
|
88
|
+
raise PacketError(f"extraction {i} has a char_interval outside the document text")
|
|
89
|
+
cand["evidence"] = {
|
|
90
|
+
"start": offsets.to_bytes(start),
|
|
91
|
+
"end": offsets.to_bytes(end),
|
|
92
|
+
"text": quote,
|
|
93
|
+
}
|
|
94
|
+
out.append(cand)
|
|
95
|
+
return out
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def admit_document(
|
|
99
|
+
document: Any,
|
|
100
|
+
schema: Schema | Mapping[str, Any],
|
|
101
|
+
policy: Policy | Mapping[str, Any] | None = None,
|
|
102
|
+
*,
|
|
103
|
+
document_id: str | None = None,
|
|
104
|
+
**options: Any,
|
|
105
|
+
) -> Receipt:
|
|
106
|
+
"""Admit every extraction in a LangExtract document. ``options`` go to ``to_candidates``.
|
|
107
|
+
|
|
108
|
+
``document_id`` defaults to the id the caller gave LangExtract. An id LangExtract generated
|
|
109
|
+
itself is random, so it is left out to keep the receipt reproducible.
|
|
110
|
+
"""
|
|
111
|
+
if document_id is None:
|
|
112
|
+
if isinstance(document, Mapping):
|
|
113
|
+
given = document.get("document_id")
|
|
114
|
+
else: # reading the property would make LangExtract invent a random id
|
|
115
|
+
given = vars(document).get("_document_id") if hasattr(document, "__dict__") else None
|
|
116
|
+
document_id = given if isinstance(given, str) else None
|
|
117
|
+
return admit(
|
|
118
|
+
document_text(document),
|
|
119
|
+
schema,
|
|
120
|
+
to_candidates(document, **options),
|
|
121
|
+
policy,
|
|
122
|
+
document_id=document_id,
|
|
123
|
+
)
|
groundgate/admit.py
ADDED
|
@@ -0,0 +1,333 @@
|
|
|
1
|
+
"""The decision procedure (SPEC §3) and receipts (SPEC §5)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections import defaultdict
|
|
6
|
+
from collections.abc import Mapping, Sequence
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from decimal import Decimal
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
from .canonical import Offsets, digest, is_nfc
|
|
12
|
+
from .model import Decision, Field, Outcome, PacketError, Policy, Receipt, Schema
|
|
13
|
+
from .text import (
|
|
14
|
+
Token,
|
|
15
|
+
canonical,
|
|
16
|
+
normalize_ws,
|
|
17
|
+
parse_value,
|
|
18
|
+
qualifiers,
|
|
19
|
+
quote_pattern,
|
|
20
|
+
scale_word,
|
|
21
|
+
tokens,
|
|
22
|
+
unit_at,
|
|
23
|
+
verbatim_equal,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
NULL_LITERALS = {"null", "none", "nil", "n/a"}
|
|
27
|
+
FLAG_ORDER = ( # SPEC §3 table order; it is part of every receipt hash
|
|
28
|
+
"NON_VERBATIM_EVIDENCE",
|
|
29
|
+
"QUALIFIED_VALUE",
|
|
30
|
+
"SCALE_WORD",
|
|
31
|
+
"LOW_CONFIDENCE",
|
|
32
|
+
"CONFLICTING_CANDIDATES",
|
|
33
|
+
)
|
|
34
|
+
REANCHORED = "EVIDENCE_REANCHORED"
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class _Reject(Exception):
|
|
38
|
+
def __init__(self, code: str) -> None:
|
|
39
|
+
self.code = code
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _span(obj: object) -> tuple[int, int] | None:
|
|
43
|
+
"""(start, end) from a span object, or None if malformed (SPEC §3 step 1)."""
|
|
44
|
+
if not isinstance(obj, Mapping):
|
|
45
|
+
return None
|
|
46
|
+
s, e = obj.get("start"), obj.get("end")
|
|
47
|
+
if (
|
|
48
|
+
isinstance(s, bool)
|
|
49
|
+
or isinstance(e, bool)
|
|
50
|
+
or not isinstance(s, int)
|
|
51
|
+
or not isinstance(e, int)
|
|
52
|
+
):
|
|
53
|
+
return None
|
|
54
|
+
return s, e
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _valid(offsets: Offsets, span: tuple[int, int]) -> tuple[int, int] | None:
|
|
58
|
+
"""Code-point span for a valid byte span (SPEC §2.2), else None."""
|
|
59
|
+
s, e = span
|
|
60
|
+
if not (0 <= s < e <= offsets.byte_length):
|
|
61
|
+
return None
|
|
62
|
+
cs, ce = offsets.to_char(s), offsets.to_char(e)
|
|
63
|
+
return None if cs is None or ce is None else (cs, ce)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
@dataclass
|
|
67
|
+
class _Ctx:
|
|
68
|
+
text: str
|
|
69
|
+
offsets: Offsets
|
|
70
|
+
schema: Schema
|
|
71
|
+
policy: Policy
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
@dataclass
|
|
75
|
+
class _Passed:
|
|
76
|
+
"""A candidate that passed steps 1-10."""
|
|
77
|
+
|
|
78
|
+
field: Field
|
|
79
|
+
value: str
|
|
80
|
+
unit: str | None
|
|
81
|
+
span: tuple[int, int] # code points
|
|
82
|
+
token: Token | None
|
|
83
|
+
flags: list[str]
|
|
84
|
+
reanchored: bool
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _value_at(
|
|
88
|
+
ctx: _Ctx, f: Field, value: Decimal | str, span: tuple[int, int]
|
|
89
|
+
) -> tuple[Token | None, str | None]:
|
|
90
|
+
"""Steps 9-10 at one span: the supporting token (None for strings) or a failure code."""
|
|
91
|
+
s, e = span
|
|
92
|
+
if f.type == "string":
|
|
93
|
+
assert isinstance(value, str)
|
|
94
|
+
if normalize_ws(value) not in normalize_ws(ctx.text[s:e]):
|
|
95
|
+
return None, "VALUE_NOT_IN_EVIDENCE"
|
|
96
|
+
return None, None
|
|
97
|
+
hits = [t for t in tokens(ctx.text, s, e) if t.value is not None and t.value == value]
|
|
98
|
+
if not hits:
|
|
99
|
+
return None, "VALUE_NOT_IN_EVIDENCE"
|
|
100
|
+
if f.unit is None:
|
|
101
|
+
return hits[0], None
|
|
102
|
+
prefixes, suffixes = ctx.schema.units.get(f.unit, ([], []))
|
|
103
|
+
for t in hits:
|
|
104
|
+
if unit_at(ctx.text, t, prefixes, suffixes, ctx.policy.unit_window):
|
|
105
|
+
return t, None
|
|
106
|
+
return None, "UNIT_NOT_IN_EVIDENCE"
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _check(ctx: _Ctx, cand: object) -> _Passed:
|
|
110
|
+
"""Steps 1-10 plus per-candidate flags. Raises _Reject."""
|
|
111
|
+
# 1. structure
|
|
112
|
+
if not isinstance(cand, Mapping):
|
|
113
|
+
raise _Reject("CANDIDATE_INVALID")
|
|
114
|
+
name, raw, unit = cand.get("field"), cand.get("value"), cand.get("unit")
|
|
115
|
+
conf, ev, region = cand.get("confidence"), cand.get("evidence"), cand.get("search_region")
|
|
116
|
+
if not isinstance(name, str) or isinstance(raw, bool) or not isinstance(raw, (str, int)):
|
|
117
|
+
raise _Reject("CANDIDATE_INVALID")
|
|
118
|
+
if unit is not None and not isinstance(unit, str):
|
|
119
|
+
raise _Reject("CANDIDATE_INVALID")
|
|
120
|
+
if conf is not None and (
|
|
121
|
+
isinstance(conf, bool) or not isinstance(conf, (int, float)) or not 0 <= conf <= 1
|
|
122
|
+
):
|
|
123
|
+
raise _Reject("CANDIDATE_INVALID")
|
|
124
|
+
cited = None
|
|
125
|
+
if ev is not None:
|
|
126
|
+
cited = _span(ev)
|
|
127
|
+
if cited is None or not isinstance(ev.get("text", ""), str):
|
|
128
|
+
raise _Reject("CANDIDATE_INVALID")
|
|
129
|
+
region_span = None
|
|
130
|
+
if region is not None:
|
|
131
|
+
region_span = _span(region)
|
|
132
|
+
if region_span is None:
|
|
133
|
+
raise _Reject("CANDIDATE_INVALID")
|
|
134
|
+
# 2-6. field, value, type, range, unit
|
|
135
|
+
f = ctx.schema.fields.get(name)
|
|
136
|
+
if f is None:
|
|
137
|
+
raise _Reject("FIELD_UNKNOWN")
|
|
138
|
+
text_value = str(raw)
|
|
139
|
+
if text_value.strip().lower() in NULL_LITERALS:
|
|
140
|
+
raise _Reject("NULL_STRING_LITERAL")
|
|
141
|
+
value: Decimal | str
|
|
142
|
+
if f.type == "string":
|
|
143
|
+
if not isinstance(raw, str) or not normalize_ws(raw):
|
|
144
|
+
raise _Reject("TYPE_INVALID")
|
|
145
|
+
value = normalize_ws(raw)
|
|
146
|
+
else:
|
|
147
|
+
parsed = parse_value(text_value)
|
|
148
|
+
if parsed is None or (f.type == "integer" and parsed != parsed.to_integral_value()):
|
|
149
|
+
raise _Reject("TYPE_INVALID")
|
|
150
|
+
value = parsed
|
|
151
|
+
if (f.minimum is not None and parsed < f.minimum) or (
|
|
152
|
+
f.maximum is not None and parsed > f.maximum
|
|
153
|
+
):
|
|
154
|
+
raise _Reject("RANGE_INVALID")
|
|
155
|
+
if unit != f.unit:
|
|
156
|
+
raise _Reject("UNIT_INVALID")
|
|
157
|
+
# 7-8. evidence
|
|
158
|
+
if cited is None:
|
|
159
|
+
raise _Reject("NO_EVIDENCE")
|
|
160
|
+
span = _valid(ctx.offsets, cited)
|
|
161
|
+
search = (0, len(ctx.text)) if region_span is None else _valid(ctx.offsets, region_span)
|
|
162
|
+
if span is None or search is None:
|
|
163
|
+
raise _Reject("SPAN_INVALID")
|
|
164
|
+
# 9-10. value and unit at the evidence, with re-anchoring
|
|
165
|
+
assert isinstance(ev, Mapping)
|
|
166
|
+
quote = ev.get("text")
|
|
167
|
+
token, failure = _value_at(ctx, f, value, span)
|
|
168
|
+
reanchored = False
|
|
169
|
+
if failure is not None:
|
|
170
|
+
if not (ctx.policy.reanchor and isinstance(quote, str) and normalize_ws(quote)):
|
|
171
|
+
raise _Reject(failure)
|
|
172
|
+
passing = []
|
|
173
|
+
for m in quote_pattern(quote).finditer(ctx.text, search[0], search[1]):
|
|
174
|
+
alt = (m.start(), m.end())
|
|
175
|
+
if alt != span:
|
|
176
|
+
alt_token, alt_failure = _value_at(ctx, f, value, alt)
|
|
177
|
+
if alt_failure is None:
|
|
178
|
+
passing.append((alt, alt_token))
|
|
179
|
+
if len(passing) != 1:
|
|
180
|
+
raise _Reject(failure)
|
|
181
|
+
(span, token), reanchored = passing[0], True
|
|
182
|
+
# flags
|
|
183
|
+
flags = []
|
|
184
|
+
if isinstance(quote, str) and not verbatim_equal(quote, ctx.text[span[0] : span[1]]):
|
|
185
|
+
flags.append("NON_VERBATIM_EVIDENCE")
|
|
186
|
+
if token is not None:
|
|
187
|
+
found = qualifiers(ctx.text, token) - {f.comparator}
|
|
188
|
+
if found:
|
|
189
|
+
flags.append("QUALIFIED_VALUE")
|
|
190
|
+
if scale_word(ctx.text, token):
|
|
191
|
+
flags.append("SCALE_WORD")
|
|
192
|
+
mc = ctx.policy.min_confidence
|
|
193
|
+
if mc is not None and conf is not None and conf < mc:
|
|
194
|
+
flags.append("LOW_CONFIDENCE")
|
|
195
|
+
canon = value if isinstance(value, str) else canonical(value)
|
|
196
|
+
return _Passed(f, canon, unit, span, token, flags, reanchored)
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def admit(
|
|
200
|
+
text: str,
|
|
201
|
+
schema: Schema | Mapping[str, Any],
|
|
202
|
+
candidates: Sequence[object],
|
|
203
|
+
policy: Policy | Mapping[str, Any] | None = None,
|
|
204
|
+
document_id: str | None = None,
|
|
205
|
+
) -> Receipt:
|
|
206
|
+
"""Decide every candidate and return the receipt."""
|
|
207
|
+
if not isinstance(text, str) or not is_nfc(text):
|
|
208
|
+
raise PacketError("document text must be a string in Unicode NFC")
|
|
209
|
+
if not isinstance(schema, Schema):
|
|
210
|
+
schema = Schema.from_dict(schema)
|
|
211
|
+
if not isinstance(policy, Policy):
|
|
212
|
+
policy = Policy() if policy is None else Policy.from_dict(policy)
|
|
213
|
+
if isinstance(candidates, (str, bytes)) or not isinstance(candidates, Sequence):
|
|
214
|
+
raise PacketError("candidates must be a list")
|
|
215
|
+
ctx = _Ctx(text, Offsets(text), schema, policy)
|
|
216
|
+
|
|
217
|
+
results: list[tuple[str, int, object, _Passed | str]] = []
|
|
218
|
+
for i, cand in enumerate(candidates):
|
|
219
|
+
try:
|
|
220
|
+
outcome: _Passed | str = _check(ctx, cand)
|
|
221
|
+
except _Reject as r:
|
|
222
|
+
outcome = r.code
|
|
223
|
+
results.append((digest("candidate", cand), i, cand, outcome))
|
|
224
|
+
|
|
225
|
+
by_field: dict[str, set[tuple[str, str | None]]] = defaultdict(set)
|
|
226
|
+
for *_, p in results:
|
|
227
|
+
if isinstance(p, _Passed) and not p.field.multiple:
|
|
228
|
+
by_field[p.field.name].add((p.value, p.unit))
|
|
229
|
+
for *_, p in results:
|
|
230
|
+
if isinstance(p, _Passed) and len(by_field.get(p.field.name, ())) > 1:
|
|
231
|
+
p.flags.append("CONFLICTING_CANDIDATES")
|
|
232
|
+
|
|
233
|
+
decisions = []
|
|
234
|
+
for sha, _, cand, p in sorted(results, key=lambda r: (r[0], r[1])):
|
|
235
|
+
cid = cand.get("id") if isinstance(cand, Mapping) else None
|
|
236
|
+
decisions.append(_decision(ctx, sha, cand, cid, p))
|
|
237
|
+
|
|
238
|
+
live = {d.field for d in decisions if d.outcome != "rejected"}
|
|
239
|
+
coverage = tuple(
|
|
240
|
+
(name, "REQUIRED_FIELD_MISSING")
|
|
241
|
+
for name in sorted(schema.fields)
|
|
242
|
+
if schema.fields[name].required and name not in live
|
|
243
|
+
)
|
|
244
|
+
doc_sha = digest("document", {"text": text})
|
|
245
|
+
draft = Receipt(
|
|
246
|
+
document_id,
|
|
247
|
+
doc_sha,
|
|
248
|
+
digest("schema", schema.to_dict()),
|
|
249
|
+
digest("policy", policy.to_dict()),
|
|
250
|
+
tuple(decisions),
|
|
251
|
+
coverage,
|
|
252
|
+
"",
|
|
253
|
+
)
|
|
254
|
+
return Receipt(**{**draft.__dict__, "receipt_sha256": digest("receipt", draft.body())})
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def _decision(ctx: _Ctx, sha: str, cand: object, cid: object, p: _Passed | str) -> Decision:
|
|
258
|
+
if isinstance(p, _Passed):
|
|
259
|
+
flags = [c for c in FLAG_ORDER if c in p.flags]
|
|
260
|
+
codes = [*flags, REANCHORED] if p.reanchored else flags
|
|
261
|
+
byte_span = (ctx.offsets.to_bytes(p.span[0]), ctx.offsets.to_bytes(p.span[1]))
|
|
262
|
+
outcome: Outcome = "needs_verification" if flags else "admitted"
|
|
263
|
+
return Decision(
|
|
264
|
+
_json_id(cid), sha, p.field.name, outcome, tuple(codes), p.value, p.unit, byte_span
|
|
265
|
+
)
|
|
266
|
+
# rejected: report what is known about the candidate
|
|
267
|
+
field_name: str | None = None
|
|
268
|
+
value: str | None = None
|
|
269
|
+
unit: str | None = None
|
|
270
|
+
span: tuple[int, int] | None = None
|
|
271
|
+
if isinstance(cand, Mapping):
|
|
272
|
+
if isinstance(cand.get("field"), str):
|
|
273
|
+
field_name = cand["field"]
|
|
274
|
+
if isinstance(cand.get("unit"), str):
|
|
275
|
+
unit = cand["unit"]
|
|
276
|
+
f = ctx.schema.fields.get(field_name or "")
|
|
277
|
+
raw = cand.get("value")
|
|
278
|
+
if f is not None and isinstance(raw, (str, int)) and not isinstance(raw, bool):
|
|
279
|
+
if f.type == "string" and isinstance(raw, str) and normalize_ws(raw):
|
|
280
|
+
value = normalize_ws(raw)
|
|
281
|
+
elif f.type != "string" and (parsed := parse_value(str(raw))) is not None:
|
|
282
|
+
value = canonical(parsed)
|
|
283
|
+
cited = _span(cand.get("evidence"))
|
|
284
|
+
if cited is not None and _valid(ctx.offsets, cited) is not None:
|
|
285
|
+
span = cited
|
|
286
|
+
return Decision(_json_id(cid), sha, field_name, "rejected", (p,), value, unit, span)
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def _json_id(cid: object) -> object:
|
|
290
|
+
return (
|
|
291
|
+
cid if cid is None or (isinstance(cid, (str, int)) and not isinstance(cid, bool)) else None
|
|
292
|
+
)
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
@dataclass(frozen=True)
|
|
296
|
+
class Verification:
|
|
297
|
+
ok: bool
|
|
298
|
+
problems: tuple[str, ...]
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def verify(
|
|
302
|
+
receipt: Mapping[str, Any],
|
|
303
|
+
text: str,
|
|
304
|
+
schema: Schema | Mapping[str, Any],
|
|
305
|
+
candidates: Sequence[object],
|
|
306
|
+
policy: Policy | Mapping[str, Any] | None = None,
|
|
307
|
+
) -> Verification:
|
|
308
|
+
"""Re-derive the receipt from its inputs and compare it with ``receipt``."""
|
|
309
|
+
if not isinstance(receipt, Mapping):
|
|
310
|
+
raise PacketError("receipt must be a JSON object")
|
|
311
|
+
problems = []
|
|
312
|
+
body = {k: v for k, v in receipt.items() if k != "receipt_sha256"}
|
|
313
|
+
if digest("receipt", body) != receipt.get("receipt_sha256"):
|
|
314
|
+
problems.append("receipt_sha256 does not match the receipt body")
|
|
315
|
+
doc_id = (
|
|
316
|
+
receipt.get("document", {}).get("id")
|
|
317
|
+
if isinstance(receipt.get("document"), Mapping)
|
|
318
|
+
else None
|
|
319
|
+
)
|
|
320
|
+
fresh = admit(text, schema, candidates, policy, document_id=doc_id).to_dict()
|
|
321
|
+
for key in ("document", "schema_sha256", "policy_sha256", "coverage", "summary"):
|
|
322
|
+
if receipt.get(key) != fresh[key]:
|
|
323
|
+
problems.append(f"{key} differs from the re-derived receipt")
|
|
324
|
+
theirs, ours = receipt.get("decisions"), fresh["decisions"]
|
|
325
|
+
if not isinstance(theirs, list) or len(theirs) != len(ours):
|
|
326
|
+
problems.append("decision count differs from the re-derived receipt")
|
|
327
|
+
else:
|
|
328
|
+
for a, b in zip(theirs, ours, strict=True):
|
|
329
|
+
if a != b:
|
|
330
|
+
problems.append(f"decision for candidate {b['candidate_sha256']} differs")
|
|
331
|
+
if not problems and receipt.get("receipt_sha256") != fresh["receipt_sha256"]:
|
|
332
|
+
problems.append("receipt_sha256 differs from the re-derived receipt") # pragma: no cover
|
|
333
|
+
return Verification(not problems, tuple(problems))
|
groundgate/canonical.py
ADDED
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
"""Canonical JSON (RFC 8785), digests, and UTF-8 byte offsets (SPEC §2.2, §6)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import math
|
|
7
|
+
import unicodedata
|
|
8
|
+
from decimal import Decimal
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
SPEC_VERSION = "0.1"
|
|
12
|
+
_SAFE_INT = 2**53 - 1
|
|
13
|
+
_ESCAPES = {
|
|
14
|
+
'"': '\\"',
|
|
15
|
+
"\\": "\\\\",
|
|
16
|
+
"\b": "\\b",
|
|
17
|
+
"\f": "\\f",
|
|
18
|
+
"\n": "\\n",
|
|
19
|
+
"\r": "\\r",
|
|
20
|
+
"\t": "\\t",
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def is_nfc(text: str) -> bool:
|
|
25
|
+
return unicodedata.is_normalized("NFC", text)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _number(x: float) -> str:
|
|
29
|
+
"""ECMAScript Number.prototype.toString for a finite float."""
|
|
30
|
+
if not math.isfinite(x):
|
|
31
|
+
raise ValueError("NaN and infinities are not permitted in canonical JSON")
|
|
32
|
+
if x == 0:
|
|
33
|
+
return "0"
|
|
34
|
+
if x < 0:
|
|
35
|
+
return "-" + _number(-x)
|
|
36
|
+
_, digits, exp = Decimal(repr(x)).as_tuple()
|
|
37
|
+
assert isinstance(exp, int)
|
|
38
|
+
ds = "".join(map(str, digits)).rstrip("0")
|
|
39
|
+
exp += len(digits) - len(ds)
|
|
40
|
+
k = len(ds)
|
|
41
|
+
n = exp + k
|
|
42
|
+
if k <= n <= 21:
|
|
43
|
+
return ds + "0" * (n - k)
|
|
44
|
+
if 0 < n <= 21:
|
|
45
|
+
return ds[:n] + "." + ds[n:]
|
|
46
|
+
if -6 < n <= 0:
|
|
47
|
+
return "0." + "0" * -n + ds
|
|
48
|
+
e = n - 1
|
|
49
|
+
mantissa = ds if k == 1 else ds[0] + "." + ds[1:]
|
|
50
|
+
return f"{mantissa}e{'+' if e >= 0 else '-'}{abs(e)}"
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _string(s: str) -> str:
|
|
54
|
+
out = ['"']
|
|
55
|
+
for ch in s:
|
|
56
|
+
o = ord(ch)
|
|
57
|
+
if ch in _ESCAPES:
|
|
58
|
+
out.append(_ESCAPES[ch])
|
|
59
|
+
elif o < 0x20 or 0xD800 <= o <= 0xDFFF:
|
|
60
|
+
out.append(f"\\u{o:04x}")
|
|
61
|
+
else:
|
|
62
|
+
out.append(ch)
|
|
63
|
+
out.append('"')
|
|
64
|
+
return "".join(out)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def jcs(obj: Any) -> str:
|
|
68
|
+
"""Serialise ``obj`` as RFC 8785 canonical JSON."""
|
|
69
|
+
if obj is None:
|
|
70
|
+
return "null"
|
|
71
|
+
if obj is True:
|
|
72
|
+
return "true"
|
|
73
|
+
if obj is False:
|
|
74
|
+
return "false"
|
|
75
|
+
if isinstance(obj, int):
|
|
76
|
+
if abs(obj) > _SAFE_INT:
|
|
77
|
+
raise ValueError(f"integer {obj} is outside the exactly representable range")
|
|
78
|
+
return str(obj)
|
|
79
|
+
if isinstance(obj, float):
|
|
80
|
+
return _number(obj)
|
|
81
|
+
if isinstance(obj, str):
|
|
82
|
+
return _string(obj)
|
|
83
|
+
if isinstance(obj, (list, tuple)):
|
|
84
|
+
return "[" + ",".join(jcs(v) for v in obj) + "]"
|
|
85
|
+
if isinstance(obj, dict):
|
|
86
|
+
for k in obj:
|
|
87
|
+
if not isinstance(k, str):
|
|
88
|
+
raise TypeError("object keys must be strings")
|
|
89
|
+
keys = sorted(obj, key=lambda k: k.encode("utf-16-be", "surrogatepass"))
|
|
90
|
+
return "{" + ",".join(_string(k) + ":" + jcs(obj[k]) for k in keys) + "}"
|
|
91
|
+
raise TypeError(f"{type(obj).__name__} is not JSON-serialisable")
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def digest(kind: str, obj: Any) -> str:
|
|
95
|
+
"""Domain-separated SHA-256 of ``obj``'s canonical JSON (SPEC §6)."""
|
|
96
|
+
prefix = f"groundgate/{SPEC_VERSION}:{kind}\0".encode("ascii")
|
|
97
|
+
payload = jcs(obj).encode("utf-8", "surrogatepass")
|
|
98
|
+
return "sha256:" + hashlib.sha256(prefix + payload).hexdigest()
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
class Offsets:
|
|
102
|
+
"""Converts between UTF-8 byte offsets and code-point offsets in one text."""
|
|
103
|
+
|
|
104
|
+
def __init__(self, text: str) -> None:
|
|
105
|
+
self.text = text
|
|
106
|
+
self._byte_at: list[int] = [0]
|
|
107
|
+
for ch in text:
|
|
108
|
+
self._byte_at.append(self._byte_at[-1] + len(ch.encode("utf-8", "surrogatepass")))
|
|
109
|
+
self._char_at = {b: i for i, b in enumerate(self._byte_at)}
|
|
110
|
+
|
|
111
|
+
@property
|
|
112
|
+
def byte_length(self) -> int:
|
|
113
|
+
return self._byte_at[-1]
|
|
114
|
+
|
|
115
|
+
def to_bytes(self, char_offset: int) -> int:
|
|
116
|
+
return self._byte_at[char_offset]
|
|
117
|
+
|
|
118
|
+
def to_char(self, byte_offset: int) -> int | None:
|
|
119
|
+
"""Code-point offset for a byte offset, or None if it is not on a character boundary."""
|
|
120
|
+
return self._char_at.get(byte_offset)
|