groundgate 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
groundgate/__init__.py ADDED
@@ -0,0 +1,22 @@
1
+ """Deterministic admission for LLM-extracted facts."""
2
+
3
+ from .admit import Verification, admit, verify
4
+ from .canonical import digest, jcs
5
+ from .model import Decision, Field, PacketError, Policy, Receipt, Schema
6
+
7
+ __version__ = "0.1.0"
8
+
9
+ __all__ = [
10
+ "Decision",
11
+ "Field",
12
+ "PacketError",
13
+ "Policy",
14
+ "Receipt",
15
+ "Schema",
16
+ "Verification",
17
+ "__version__",
18
+ "admit",
19
+ "digest",
20
+ "jcs",
21
+ "verify",
22
+ ]
@@ -0,0 +1 @@
1
+ """Adapters from other extraction tools' output to groundgate candidates."""
@@ -0,0 +1,123 @@
1
+ """Admit LangExtract results.
2
+
3
+ LangExtract checks that each ``extraction_text`` is found in the source and records where
4
+ (``char_interval``). It does not check the typed value you asked for in the attributes. This
5
+ adapter turns each extraction into a groundgate candidate so the value and unit are checked at
6
+ that location::
7
+
8
+ result = lx.extract(text_or_documents=text, prompt_description=..., examples=...)
9
+ receipt = admit_document(result, schema)
10
+
11
+ Works on ``lx.data.AnnotatedDocument`` objects and on the dicts LangExtract writes to JSONL
12
+ (``lx.io.save_annotated_documents``). LangExtract itself is not imported.
13
+
14
+ Mapping, per extraction:
15
+
16
+ - ``field``: ``extraction_class``, renamed through ``fields`` when given;
17
+ - ``value``: the ``value`` attribute, or ``extraction_text`` when there is none;
18
+ - ``unit``: the ``unit`` attribute, when present;
19
+ - ``evidence``: ``char_interval`` converted to UTF-8 byte offsets, quoting ``extraction_text``.
20
+ An extraction LangExtract could not align has no evidence and is rejected ``NO_EVIDENCE``.
21
+
22
+ ``alignment_status`` and ``extraction_class`` are kept on the candidate for the report; the
23
+ checks ignore them.
24
+ """
25
+
26
+ from __future__ import annotations
27
+
28
+ from collections.abc import Mapping
29
+ from typing import Any
30
+
31
+ from ..admit import admit
32
+ from ..canonical import Offsets, is_nfc
33
+ from ..model import PacketError, Policy, Receipt, Schema
34
+
35
+
36
+ def _get(obj: Any, name: str) -> Any:
37
+ if isinstance(obj, Mapping):
38
+ return obj.get(name)
39
+ return getattr(obj, name, None)
40
+
41
+
42
+ def _status(raw: Any) -> str | None:
43
+ if raw is None:
44
+ return None
45
+ return str(getattr(raw, "value", raw)) # an AlignmentStatus enum or its string value
46
+
47
+
48
+ def document_text(document: Any) -> str:
49
+ text = _get(document, "text")
50
+ if not isinstance(text, str):
51
+ raise PacketError("the LangExtract document has no text")
52
+ if not is_nfc(text):
53
+ raise PacketError(
54
+ "the LangExtract document text is not NFC; normalise it before calling lx.extract "
55
+ "(unicodedata.normalize('NFC', text)) so offsets line up"
56
+ )
57
+ return text
58
+
59
+
60
+ def to_candidates(
61
+ document: Any,
62
+ *,
63
+ fields: Mapping[str, str] | None = None,
64
+ value_attribute: str = "value",
65
+ unit_attribute: str = "unit",
66
+ ) -> list[dict[str, Any]]:
67
+ """One groundgate candidate per extraction in a LangExtract ``AnnotatedDocument``."""
68
+ text = document_text(document)
69
+ offsets = Offsets(text)
70
+ out = []
71
+ for i, x in enumerate(_get(document, "extractions") or []):
72
+ cls = _get(x, "extraction_class")
73
+ quote = _get(x, "extraction_text")
74
+ attrs = _get(x, "attributes") or {}
75
+ cand: dict[str, Any] = {
76
+ "id": i,
77
+ "field": (fields or {}).get(cls, cls),
78
+ "value": attrs.get(value_attribute, quote),
79
+ "extraction_class": cls,
80
+ "alignment_status": _status(_get(x, "alignment_status")),
81
+ }
82
+ if attrs.get(unit_attribute) is not None:
83
+ cand["unit"] = attrs[unit_attribute]
84
+ interval = _get(x, "char_interval")
85
+ start, end = _get(interval, "start_pos"), _get(interval, "end_pos")
86
+ if isinstance(start, int) and isinstance(end, int):
87
+ if not 0 <= start <= end <= len(text):
88
+ raise PacketError(f"extraction {i} has a char_interval outside the document text")
89
+ cand["evidence"] = {
90
+ "start": offsets.to_bytes(start),
91
+ "end": offsets.to_bytes(end),
92
+ "text": quote,
93
+ }
94
+ out.append(cand)
95
+ return out
96
+
97
+
98
+ def admit_document(
99
+ document: Any,
100
+ schema: Schema | Mapping[str, Any],
101
+ policy: Policy | Mapping[str, Any] | None = None,
102
+ *,
103
+ document_id: str | None = None,
104
+ **options: Any,
105
+ ) -> Receipt:
106
+ """Admit every extraction in a LangExtract document. ``options`` go to ``to_candidates``.
107
+
108
+ ``document_id`` defaults to the id the caller gave LangExtract. An id LangExtract generated
109
+ itself is random, so it is left out to keep the receipt reproducible.
110
+ """
111
+ if document_id is None:
112
+ if isinstance(document, Mapping):
113
+ given = document.get("document_id")
114
+ else: # reading the property would make LangExtract invent a random id
115
+ given = vars(document).get("_document_id") if hasattr(document, "__dict__") else None
116
+ document_id = given if isinstance(given, str) else None
117
+ return admit(
118
+ document_text(document),
119
+ schema,
120
+ to_candidates(document, **options),
121
+ policy,
122
+ document_id=document_id,
123
+ )
groundgate/admit.py ADDED
@@ -0,0 +1,333 @@
1
+ """The decision procedure (SPEC §3) and receipts (SPEC §5)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections import defaultdict
6
+ from collections.abc import Mapping, Sequence
7
+ from dataclasses import dataclass
8
+ from decimal import Decimal
9
+ from typing import Any
10
+
11
+ from .canonical import Offsets, digest, is_nfc
12
+ from .model import Decision, Field, Outcome, PacketError, Policy, Receipt, Schema
13
+ from .text import (
14
+ Token,
15
+ canonical,
16
+ normalize_ws,
17
+ parse_value,
18
+ qualifiers,
19
+ quote_pattern,
20
+ scale_word,
21
+ tokens,
22
+ unit_at,
23
+ verbatim_equal,
24
+ )
25
+
26
+ NULL_LITERALS = {"null", "none", "nil", "n/a"}
27
+ FLAG_ORDER = ( # SPEC §3 table order; it is part of every receipt hash
28
+ "NON_VERBATIM_EVIDENCE",
29
+ "QUALIFIED_VALUE",
30
+ "SCALE_WORD",
31
+ "LOW_CONFIDENCE",
32
+ "CONFLICTING_CANDIDATES",
33
+ )
34
+ REANCHORED = "EVIDENCE_REANCHORED"
35
+
36
+
37
+ class _Reject(Exception):
38
+ def __init__(self, code: str) -> None:
39
+ self.code = code
40
+
41
+
42
+ def _span(obj: object) -> tuple[int, int] | None:
43
+ """(start, end) from a span object, or None if malformed (SPEC §3 step 1)."""
44
+ if not isinstance(obj, Mapping):
45
+ return None
46
+ s, e = obj.get("start"), obj.get("end")
47
+ if (
48
+ isinstance(s, bool)
49
+ or isinstance(e, bool)
50
+ or not isinstance(s, int)
51
+ or not isinstance(e, int)
52
+ ):
53
+ return None
54
+ return s, e
55
+
56
+
57
+ def _valid(offsets: Offsets, span: tuple[int, int]) -> tuple[int, int] | None:
58
+ """Code-point span for a valid byte span (SPEC §2.2), else None."""
59
+ s, e = span
60
+ if not (0 <= s < e <= offsets.byte_length):
61
+ return None
62
+ cs, ce = offsets.to_char(s), offsets.to_char(e)
63
+ return None if cs is None or ce is None else (cs, ce)
64
+
65
+
66
+ @dataclass
67
+ class _Ctx:
68
+ text: str
69
+ offsets: Offsets
70
+ schema: Schema
71
+ policy: Policy
72
+
73
+
74
+ @dataclass
75
+ class _Passed:
76
+ """A candidate that passed steps 1-10."""
77
+
78
+ field: Field
79
+ value: str
80
+ unit: str | None
81
+ span: tuple[int, int] # code points
82
+ token: Token | None
83
+ flags: list[str]
84
+ reanchored: bool
85
+
86
+
87
+ def _value_at(
88
+ ctx: _Ctx, f: Field, value: Decimal | str, span: tuple[int, int]
89
+ ) -> tuple[Token | None, str | None]:
90
+ """Steps 9-10 at one span: the supporting token (None for strings) or a failure code."""
91
+ s, e = span
92
+ if f.type == "string":
93
+ assert isinstance(value, str)
94
+ if normalize_ws(value) not in normalize_ws(ctx.text[s:e]):
95
+ return None, "VALUE_NOT_IN_EVIDENCE"
96
+ return None, None
97
+ hits = [t for t in tokens(ctx.text, s, e) if t.value is not None and t.value == value]
98
+ if not hits:
99
+ return None, "VALUE_NOT_IN_EVIDENCE"
100
+ if f.unit is None:
101
+ return hits[0], None
102
+ prefixes, suffixes = ctx.schema.units.get(f.unit, ([], []))
103
+ for t in hits:
104
+ if unit_at(ctx.text, t, prefixes, suffixes, ctx.policy.unit_window):
105
+ return t, None
106
+ return None, "UNIT_NOT_IN_EVIDENCE"
107
+
108
+
109
+ def _check(ctx: _Ctx, cand: object) -> _Passed:
110
+ """Steps 1-10 plus per-candidate flags. Raises _Reject."""
111
+ # 1. structure
112
+ if not isinstance(cand, Mapping):
113
+ raise _Reject("CANDIDATE_INVALID")
114
+ name, raw, unit = cand.get("field"), cand.get("value"), cand.get("unit")
115
+ conf, ev, region = cand.get("confidence"), cand.get("evidence"), cand.get("search_region")
116
+ if not isinstance(name, str) or isinstance(raw, bool) or not isinstance(raw, (str, int)):
117
+ raise _Reject("CANDIDATE_INVALID")
118
+ if unit is not None and not isinstance(unit, str):
119
+ raise _Reject("CANDIDATE_INVALID")
120
+ if conf is not None and (
121
+ isinstance(conf, bool) or not isinstance(conf, (int, float)) or not 0 <= conf <= 1
122
+ ):
123
+ raise _Reject("CANDIDATE_INVALID")
124
+ cited = None
125
+ if ev is not None:
126
+ cited = _span(ev)
127
+ if cited is None or not isinstance(ev.get("text", ""), str):
128
+ raise _Reject("CANDIDATE_INVALID")
129
+ region_span = None
130
+ if region is not None:
131
+ region_span = _span(region)
132
+ if region_span is None:
133
+ raise _Reject("CANDIDATE_INVALID")
134
+ # 2-6. field, value, type, range, unit
135
+ f = ctx.schema.fields.get(name)
136
+ if f is None:
137
+ raise _Reject("FIELD_UNKNOWN")
138
+ text_value = str(raw)
139
+ if text_value.strip().lower() in NULL_LITERALS:
140
+ raise _Reject("NULL_STRING_LITERAL")
141
+ value: Decimal | str
142
+ if f.type == "string":
143
+ if not isinstance(raw, str) or not normalize_ws(raw):
144
+ raise _Reject("TYPE_INVALID")
145
+ value = normalize_ws(raw)
146
+ else:
147
+ parsed = parse_value(text_value)
148
+ if parsed is None or (f.type == "integer" and parsed != parsed.to_integral_value()):
149
+ raise _Reject("TYPE_INVALID")
150
+ value = parsed
151
+ if (f.minimum is not None and parsed < f.minimum) or (
152
+ f.maximum is not None and parsed > f.maximum
153
+ ):
154
+ raise _Reject("RANGE_INVALID")
155
+ if unit != f.unit:
156
+ raise _Reject("UNIT_INVALID")
157
+ # 7-8. evidence
158
+ if cited is None:
159
+ raise _Reject("NO_EVIDENCE")
160
+ span = _valid(ctx.offsets, cited)
161
+ search = (0, len(ctx.text)) if region_span is None else _valid(ctx.offsets, region_span)
162
+ if span is None or search is None:
163
+ raise _Reject("SPAN_INVALID")
164
+ # 9-10. value and unit at the evidence, with re-anchoring
165
+ assert isinstance(ev, Mapping)
166
+ quote = ev.get("text")
167
+ token, failure = _value_at(ctx, f, value, span)
168
+ reanchored = False
169
+ if failure is not None:
170
+ if not (ctx.policy.reanchor and isinstance(quote, str) and normalize_ws(quote)):
171
+ raise _Reject(failure)
172
+ passing = []
173
+ for m in quote_pattern(quote).finditer(ctx.text, search[0], search[1]):
174
+ alt = (m.start(), m.end())
175
+ if alt != span:
176
+ alt_token, alt_failure = _value_at(ctx, f, value, alt)
177
+ if alt_failure is None:
178
+ passing.append((alt, alt_token))
179
+ if len(passing) != 1:
180
+ raise _Reject(failure)
181
+ (span, token), reanchored = passing[0], True
182
+ # flags
183
+ flags = []
184
+ if isinstance(quote, str) and not verbatim_equal(quote, ctx.text[span[0] : span[1]]):
185
+ flags.append("NON_VERBATIM_EVIDENCE")
186
+ if token is not None:
187
+ found = qualifiers(ctx.text, token) - {f.comparator}
188
+ if found:
189
+ flags.append("QUALIFIED_VALUE")
190
+ if scale_word(ctx.text, token):
191
+ flags.append("SCALE_WORD")
192
+ mc = ctx.policy.min_confidence
193
+ if mc is not None and conf is not None and conf < mc:
194
+ flags.append("LOW_CONFIDENCE")
195
+ canon = value if isinstance(value, str) else canonical(value)
196
+ return _Passed(f, canon, unit, span, token, flags, reanchored)
197
+
198
+
199
+ def admit(
200
+ text: str,
201
+ schema: Schema | Mapping[str, Any],
202
+ candidates: Sequence[object],
203
+ policy: Policy | Mapping[str, Any] | None = None,
204
+ document_id: str | None = None,
205
+ ) -> Receipt:
206
+ """Decide every candidate and return the receipt."""
207
+ if not isinstance(text, str) or not is_nfc(text):
208
+ raise PacketError("document text must be a string in Unicode NFC")
209
+ if not isinstance(schema, Schema):
210
+ schema = Schema.from_dict(schema)
211
+ if not isinstance(policy, Policy):
212
+ policy = Policy() if policy is None else Policy.from_dict(policy)
213
+ if isinstance(candidates, (str, bytes)) or not isinstance(candidates, Sequence):
214
+ raise PacketError("candidates must be a list")
215
+ ctx = _Ctx(text, Offsets(text), schema, policy)
216
+
217
+ results: list[tuple[str, int, object, _Passed | str]] = []
218
+ for i, cand in enumerate(candidates):
219
+ try:
220
+ outcome: _Passed | str = _check(ctx, cand)
221
+ except _Reject as r:
222
+ outcome = r.code
223
+ results.append((digest("candidate", cand), i, cand, outcome))
224
+
225
+ by_field: dict[str, set[tuple[str, str | None]]] = defaultdict(set)
226
+ for *_, p in results:
227
+ if isinstance(p, _Passed) and not p.field.multiple:
228
+ by_field[p.field.name].add((p.value, p.unit))
229
+ for *_, p in results:
230
+ if isinstance(p, _Passed) and len(by_field.get(p.field.name, ())) > 1:
231
+ p.flags.append("CONFLICTING_CANDIDATES")
232
+
233
+ decisions = []
234
+ for sha, _, cand, p in sorted(results, key=lambda r: (r[0], r[1])):
235
+ cid = cand.get("id") if isinstance(cand, Mapping) else None
236
+ decisions.append(_decision(ctx, sha, cand, cid, p))
237
+
238
+ live = {d.field for d in decisions if d.outcome != "rejected"}
239
+ coverage = tuple(
240
+ (name, "REQUIRED_FIELD_MISSING")
241
+ for name in sorted(schema.fields)
242
+ if schema.fields[name].required and name not in live
243
+ )
244
+ doc_sha = digest("document", {"text": text})
245
+ draft = Receipt(
246
+ document_id,
247
+ doc_sha,
248
+ digest("schema", schema.to_dict()),
249
+ digest("policy", policy.to_dict()),
250
+ tuple(decisions),
251
+ coverage,
252
+ "",
253
+ )
254
+ return Receipt(**{**draft.__dict__, "receipt_sha256": digest("receipt", draft.body())})
255
+
256
+
257
+ def _decision(ctx: _Ctx, sha: str, cand: object, cid: object, p: _Passed | str) -> Decision:
258
+ if isinstance(p, _Passed):
259
+ flags = [c for c in FLAG_ORDER if c in p.flags]
260
+ codes = [*flags, REANCHORED] if p.reanchored else flags
261
+ byte_span = (ctx.offsets.to_bytes(p.span[0]), ctx.offsets.to_bytes(p.span[1]))
262
+ outcome: Outcome = "needs_verification" if flags else "admitted"
263
+ return Decision(
264
+ _json_id(cid), sha, p.field.name, outcome, tuple(codes), p.value, p.unit, byte_span
265
+ )
266
+ # rejected: report what is known about the candidate
267
+ field_name: str | None = None
268
+ value: str | None = None
269
+ unit: str | None = None
270
+ span: tuple[int, int] | None = None
271
+ if isinstance(cand, Mapping):
272
+ if isinstance(cand.get("field"), str):
273
+ field_name = cand["field"]
274
+ if isinstance(cand.get("unit"), str):
275
+ unit = cand["unit"]
276
+ f = ctx.schema.fields.get(field_name or "")
277
+ raw = cand.get("value")
278
+ if f is not None and isinstance(raw, (str, int)) and not isinstance(raw, bool):
279
+ if f.type == "string" and isinstance(raw, str) and normalize_ws(raw):
280
+ value = normalize_ws(raw)
281
+ elif f.type != "string" and (parsed := parse_value(str(raw))) is not None:
282
+ value = canonical(parsed)
283
+ cited = _span(cand.get("evidence"))
284
+ if cited is not None and _valid(ctx.offsets, cited) is not None:
285
+ span = cited
286
+ return Decision(_json_id(cid), sha, field_name, "rejected", (p,), value, unit, span)
287
+
288
+
289
+ def _json_id(cid: object) -> object:
290
+ return (
291
+ cid if cid is None or (isinstance(cid, (str, int)) and not isinstance(cid, bool)) else None
292
+ )
293
+
294
+
295
+ @dataclass(frozen=True)
296
+ class Verification:
297
+ ok: bool
298
+ problems: tuple[str, ...]
299
+
300
+
301
+ def verify(
302
+ receipt: Mapping[str, Any],
303
+ text: str,
304
+ schema: Schema | Mapping[str, Any],
305
+ candidates: Sequence[object],
306
+ policy: Policy | Mapping[str, Any] | None = None,
307
+ ) -> Verification:
308
+ """Re-derive the receipt from its inputs and compare it with ``receipt``."""
309
+ if not isinstance(receipt, Mapping):
310
+ raise PacketError("receipt must be a JSON object")
311
+ problems = []
312
+ body = {k: v for k, v in receipt.items() if k != "receipt_sha256"}
313
+ if digest("receipt", body) != receipt.get("receipt_sha256"):
314
+ problems.append("receipt_sha256 does not match the receipt body")
315
+ doc_id = (
316
+ receipt.get("document", {}).get("id")
317
+ if isinstance(receipt.get("document"), Mapping)
318
+ else None
319
+ )
320
+ fresh = admit(text, schema, candidates, policy, document_id=doc_id).to_dict()
321
+ for key in ("document", "schema_sha256", "policy_sha256", "coverage", "summary"):
322
+ if receipt.get(key) != fresh[key]:
323
+ problems.append(f"{key} differs from the re-derived receipt")
324
+ theirs, ours = receipt.get("decisions"), fresh["decisions"]
325
+ if not isinstance(theirs, list) or len(theirs) != len(ours):
326
+ problems.append("decision count differs from the re-derived receipt")
327
+ else:
328
+ for a, b in zip(theirs, ours, strict=True):
329
+ if a != b:
330
+ problems.append(f"decision for candidate {b['candidate_sha256']} differs")
331
+ if not problems and receipt.get("receipt_sha256") != fresh["receipt_sha256"]:
332
+ problems.append("receipt_sha256 differs from the re-derived receipt") # pragma: no cover
333
+ return Verification(not problems, tuple(problems))
@@ -0,0 +1,120 @@
1
+ """Canonical JSON (RFC 8785), digests, and UTF-8 byte offsets (SPEC §2.2, §6)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import math
7
+ import unicodedata
8
+ from decimal import Decimal
9
+ from typing import Any
10
+
11
+ SPEC_VERSION = "0.1"
12
+ _SAFE_INT = 2**53 - 1
13
+ _ESCAPES = {
14
+ '"': '\\"',
15
+ "\\": "\\\\",
16
+ "\b": "\\b",
17
+ "\f": "\\f",
18
+ "\n": "\\n",
19
+ "\r": "\\r",
20
+ "\t": "\\t",
21
+ }
22
+
23
+
24
+ def is_nfc(text: str) -> bool:
25
+ return unicodedata.is_normalized("NFC", text)
26
+
27
+
28
+ def _number(x: float) -> str:
29
+ """ECMAScript Number.prototype.toString for a finite float."""
30
+ if not math.isfinite(x):
31
+ raise ValueError("NaN and infinities are not permitted in canonical JSON")
32
+ if x == 0:
33
+ return "0"
34
+ if x < 0:
35
+ return "-" + _number(-x)
36
+ _, digits, exp = Decimal(repr(x)).as_tuple()
37
+ assert isinstance(exp, int)
38
+ ds = "".join(map(str, digits)).rstrip("0")
39
+ exp += len(digits) - len(ds)
40
+ k = len(ds)
41
+ n = exp + k
42
+ if k <= n <= 21:
43
+ return ds + "0" * (n - k)
44
+ if 0 < n <= 21:
45
+ return ds[:n] + "." + ds[n:]
46
+ if -6 < n <= 0:
47
+ return "0." + "0" * -n + ds
48
+ e = n - 1
49
+ mantissa = ds if k == 1 else ds[0] + "." + ds[1:]
50
+ return f"{mantissa}e{'+' if e >= 0 else '-'}{abs(e)}"
51
+
52
+
53
+ def _string(s: str) -> str:
54
+ out = ['"']
55
+ for ch in s:
56
+ o = ord(ch)
57
+ if ch in _ESCAPES:
58
+ out.append(_ESCAPES[ch])
59
+ elif o < 0x20 or 0xD800 <= o <= 0xDFFF:
60
+ out.append(f"\\u{o:04x}")
61
+ else:
62
+ out.append(ch)
63
+ out.append('"')
64
+ return "".join(out)
65
+
66
+
67
+ def jcs(obj: Any) -> str:
68
+ """Serialise ``obj`` as RFC 8785 canonical JSON."""
69
+ if obj is None:
70
+ return "null"
71
+ if obj is True:
72
+ return "true"
73
+ if obj is False:
74
+ return "false"
75
+ if isinstance(obj, int):
76
+ if abs(obj) > _SAFE_INT:
77
+ raise ValueError(f"integer {obj} is outside the exactly representable range")
78
+ return str(obj)
79
+ if isinstance(obj, float):
80
+ return _number(obj)
81
+ if isinstance(obj, str):
82
+ return _string(obj)
83
+ if isinstance(obj, (list, tuple)):
84
+ return "[" + ",".join(jcs(v) for v in obj) + "]"
85
+ if isinstance(obj, dict):
86
+ for k in obj:
87
+ if not isinstance(k, str):
88
+ raise TypeError("object keys must be strings")
89
+ keys = sorted(obj, key=lambda k: k.encode("utf-16-be", "surrogatepass"))
90
+ return "{" + ",".join(_string(k) + ":" + jcs(obj[k]) for k in keys) + "}"
91
+ raise TypeError(f"{type(obj).__name__} is not JSON-serialisable")
92
+
93
+
94
+ def digest(kind: str, obj: Any) -> str:
95
+ """Domain-separated SHA-256 of ``obj``'s canonical JSON (SPEC §6)."""
96
+ prefix = f"groundgate/{SPEC_VERSION}:{kind}\0".encode("ascii")
97
+ payload = jcs(obj).encode("utf-8", "surrogatepass")
98
+ return "sha256:" + hashlib.sha256(prefix + payload).hexdigest()
99
+
100
+
101
+ class Offsets:
102
+ """Converts between UTF-8 byte offsets and code-point offsets in one text."""
103
+
104
+ def __init__(self, text: str) -> None:
105
+ self.text = text
106
+ self._byte_at: list[int] = [0]
107
+ for ch in text:
108
+ self._byte_at.append(self._byte_at[-1] + len(ch.encode("utf-8", "surrogatepass")))
109
+ self._char_at = {b: i for i, b in enumerate(self._byte_at)}
110
+
111
+ @property
112
+ def byte_length(self) -> int:
113
+ return self._byte_at[-1]
114
+
115
+ def to_bytes(self, char_offset: int) -> int:
116
+ return self._byte_at[char_offset]
117
+
118
+ def to_char(self, byte_offset: int) -> int | None:
119
+ """Code-point offset for a byte offset, or None if it is not on a character boundary."""
120
+ return self._char_at.get(byte_offset)