@hraness/message-like-me 0.8.0 → 0.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,477 +0,0 @@
1
- #!/usr/bin/env python3
2
- """Validate an Ensoul source packet without printing evidence content."""
3
-
4
- from __future__ import annotations
5
-
6
- import argparse
7
- import datetime as dt
8
- import json
9
- import os
10
- from pathlib import Path
11
- import re
12
- import sys
13
- from typing import NoReturn
14
- from urllib.parse import urlsplit
15
-
16
- sys.dont_write_bytecode = True
17
- from prepare_x_archive import ArchiveError, canonical_bytes, sha256_hex
18
-
19
-
20
- MAX_PACKET_BYTES = 128 * 1024 * 1024
21
- SAFE_INTEGER = 9_007_199_254_740_991
22
- SHA256 = re.compile(r"^[a-f0-9]{64}$")
23
- PREFIXED_SHA256 = re.compile(r"^sha256:[a-f0-9]{64}$")
24
- DATE_TIME = re.compile(
25
- r"^[0-9]{4}-[0-9]{2}-[0-9]{2}T[0-9]{2}:[0-9]{2}:[0-9]{2}(?:\.[0-9]+)?(?:Z|[+-][0-9]{2}:[0-9]{2})$"
26
- )
27
-
28
-
29
- class PacketValidationError(ValueError):
30
- pass
31
-
32
-
33
- def fail(path: str, message: str) -> NoReturn:
34
- raise PacketValidationError(f"{path}: {message}")
35
-
36
-
37
- def expect_dict(value: object, path: str) -> dict[str, object]:
38
- if not isinstance(value, dict):
39
- fail(path, "must be an object")
40
- return value
41
-
42
-
43
- def expect_list(value: object, path: str) -> list[object]:
44
- if not isinstance(value, list):
45
- fail(path, "must be an array")
46
- return value
47
-
48
-
49
- def expect_string(value: object, path: str, minimum: int, maximum: int) -> str:
50
- if not isinstance(value, str):
51
- fail(path, "must be a string")
52
- if len(value) < minimum or len(value) > maximum:
53
- fail(path, "has an invalid length")
54
- return value
55
-
56
-
57
- def expect_bool(value: object, path: str) -> bool:
58
- if not isinstance(value, bool):
59
- fail(path, "must be a boolean")
60
- return value
61
-
62
-
63
- def exact_keys(
64
- value: dict[str, object],
65
- path: str,
66
- required: set[str],
67
- optional: set[str] = frozenset(),
68
- ) -> None:
69
- missing = required - value.keys()
70
- extra = value.keys() - required - optional
71
- if missing:
72
- fail(path, f"missing required member {sorted(missing)[0]}")
73
- if extra:
74
- fail(path, f"contains unknown member {sorted(extra)[0]}")
75
-
76
-
77
- def expect_enum(value: object, path: str, allowed: set[str]) -> str:
78
- text = expect_string(value, path, 1, 200)
79
- if text not in allowed:
80
- fail(path, "has an unknown enum value")
81
- return text
82
-
83
-
84
- def parse_datetime(value: object, path: str) -> dt.datetime:
85
- text = expect_string(value, path, 1, 100)
86
- if DATE_TIME.fullmatch(text) is None:
87
- fail(path, "must be an RFC 3339 date-time")
88
- try:
89
- parsed = dt.datetime.fromisoformat(text.replace("Z", "+00:00"))
90
- except ValueError:
91
- fail(path, "must be a valid date-time")
92
- if parsed.tzinfo is None:
93
- fail(path, "must include a timezone")
94
- return parsed.astimezone(dt.timezone.utc)
95
-
96
-
97
- def expect_digest(value: object, path: str, prefixed: bool) -> str:
98
- text = expect_string(value, path, 64 if not prefixed else 71, 64 if not prefixed else 71)
99
- pattern = PREFIXED_SHA256 if prefixed else SHA256
100
- if pattern.fullmatch(text) is None:
101
- fail(path, "must be a lowercase SHA-256 digest")
102
- return text
103
-
104
-
105
- def reject_non_ijson(value: object, path: str = "packet") -> None:
106
- if value is None or isinstance(value, bool):
107
- return
108
- if isinstance(value, int):
109
- if abs(value) > SAFE_INTEGER:
110
- fail(path, "integer exceeds the interoperable JSON range")
111
- return
112
- if isinstance(value, float):
113
- fail(path, "floating-point values are not allowed by this packet schema")
114
- if isinstance(value, str):
115
- if any(0xD800 <= ord(character) <= 0xDFFF for character in value):
116
- fail(path, "contains an unpaired Unicode surrogate")
117
- return
118
- if isinstance(value, list):
119
- for index, member in enumerate(value):
120
- reject_non_ijson(member, f"{path}[{index}]")
121
- return
122
- if isinstance(value, dict):
123
- for key, member in value.items():
124
- reject_non_ijson(key, f"{path}.key")
125
- reject_non_ijson(member, f"{path}.{key}")
126
- return
127
- fail(path, "contains a non-JSON value")
128
-
129
-
130
- def strict_json_loads(data: bytes) -> object:
131
- if data.startswith(b"\xef\xbb\xbf"):
132
- fail("packet", "UTF-8 BOM is not allowed")
133
- try:
134
- text = data.decode("utf-8")
135
- except UnicodeDecodeError:
136
- fail("packet", "must be valid UTF-8")
137
-
138
- def object_pairs(pairs: list[tuple[str, object]]) -> dict[str, object]:
139
- result: dict[str, object] = {}
140
- for key, value in pairs:
141
- if key in result:
142
- fail("packet", "contains a duplicate object member")
143
- result[key] = value
144
- return result
145
-
146
- def reject_constant(_: str) -> NoReturn:
147
- fail("packet", "contains a non-finite number")
148
-
149
- def reject_float(_: str) -> NoReturn:
150
- fail("packet", "floating-point numbers are not allowed")
151
-
152
- def parse_integer(value: str) -> int:
153
- parsed = int(value)
154
- if abs(parsed) > SAFE_INTEGER:
155
- fail("packet", "integer exceeds the interoperable JSON range")
156
- return parsed
157
-
158
- try:
159
- value = json.loads(
160
- text,
161
- object_pairs_hook=object_pairs,
162
- parse_constant=reject_constant,
163
- parse_float=reject_float,
164
- parse_int=parse_integer,
165
- )
166
- except json.JSONDecodeError:
167
- fail("packet", "is not valid JSON")
168
- reject_non_ijson(value)
169
- return value
170
-
171
-
172
- def validate_subject(value: object) -> dict[str, object]:
173
- subject = expect_dict(value, "subject")
174
- exact_keys(subject, "subject", {"localId", "kind", "identityBasis"}, {"displayName"})
175
- expect_string(subject["localId"], "subject.localId", 1, 200)
176
- expect_enum(subject["kind"], "subject.kind", {"owner", "contact", "person"})
177
- expect_string(subject["identityBasis"], "subject.identityBasis", 1, 1000)
178
- if "displayName" in subject:
179
- expect_string(subject["displayName"], "subject.displayName", 1, 300)
180
- return subject
181
-
182
-
183
- def effective_bounds(limits: dict[str, object]) -> tuple[dt.datetime | None, dt.datetime | None]:
184
- direct_after = limits.get("after") if isinstance(limits.get("after"), str) else None
185
- alias_after = limits.get("afterInclusive") if isinstance(limits.get("afterInclusive"), str) else None
186
- direct_before = limits.get("before") if isinstance(limits.get("before"), str) else None
187
- alias_before = limits.get("beforeExclusive") if isinstance(limits.get("beforeExclusive"), str) else None
188
- if direct_after is not None and alias_after is not None:
189
- fail("scope.limits", "declares two lower-bound aliases")
190
- if direct_before is not None and alias_before is not None:
191
- fail("scope.limits", "declares two upper-bound aliases")
192
- lower_raw = direct_after if direct_after is not None else alias_after
193
- upper_raw = direct_before if direct_before is not None else alias_before
194
- lower = parse_datetime(lower_raw, "scope.limits lower bound") if lower_raw is not None else None
195
- upper = parse_datetime(upper_raw, "scope.limits upper bound") if upper_raw is not None else None
196
- if lower is not None and upper is not None and lower >= upper:
197
- fail("scope.limits", "lower bound must be earlier than upper bound")
198
- return lower, upper
199
-
200
-
201
- def validate_limits(value: object) -> dict[str, object]:
202
- limits = expect_dict(value, "scope.limits")
203
- if len(limits) > 32:
204
- fail("scope.limits", "has too many members")
205
- for key, member in limits.items():
206
- expect_string(key, "scope.limits key", 1, 200)
207
- if isinstance(member, list):
208
- if len(member) > 32:
209
- fail(f"scope.limits.{key}", "has too many array items")
210
- values = [expect_string(item, f"scope.limits.{key}[]", 1, 200) for item in member]
211
- if len(values) != len(set(values)):
212
- fail(f"scope.limits.{key}", "contains duplicate array items")
213
- elif member is not None and not isinstance(member, (str, int, bool)):
214
- fail(f"scope.limits.{key}", "has a disallowed value type")
215
- reject_non_ijson(member, f"scope.limits.{key}")
216
- for name in ("after", "afterInclusive", "before", "beforeExclusive"):
217
- if name in limits and isinstance(limits[name], str):
218
- parse_datetime(limits[name], f"scope.limits.{name}")
219
- effective_bounds(limits)
220
- return limits
221
-
222
-
223
- def validate_scope(
224
- value: object,
225
- ) -> tuple[dict[str, object], dict[str, object], dt.datetime | None, dt.datetime | None]:
226
- scope = expect_dict(value, "scope")
227
- exact_keys(
228
- scope,
229
- "scope",
230
- {"adapter", "payloadSchema", "completeness", "limits"},
231
- {"asOf", "sourceCutoff", "sourceRevision"},
232
- )
233
- expect_string(scope["adapter"], "scope.adapter", 1, 100)
234
- expect_string(scope["payloadSchema"], "scope.payloadSchema", 1, 160)
235
- expect_enum(scope["completeness"], "scope.completeness", {"complete", "sampled", "bounded", "unknown"})
236
- as_of = parse_datetime(scope["asOf"], "scope.asOf") if "asOf" in scope else None
237
- source_cutoff = (
238
- parse_datetime(scope["sourceCutoff"], "scope.sourceCutoff")
239
- if "sourceCutoff" in scope
240
- else None
241
- )
242
- if "sourceRevision" in scope:
243
- expect_string(scope["sourceRevision"], "scope.sourceRevision", 1, 300)
244
- return scope, validate_limits(scope["limits"]), as_of, source_cutoff
245
-
246
-
247
- def validate_content(value: object, path: str) -> dict[str, object]:
248
- content = expect_dict(value, path)
249
- exact_keys(content, path, set(), {"text", "title", "url", "truncated"})
250
- if not any(key in content for key in ("text", "title", "url")):
251
- fail(path, "must include text, title, or url")
252
- if "text" in content:
253
- expect_string(content["text"], f"{path}.text", 0, 50_000)
254
- if "title" in content:
255
- expect_string(content["title"], f"{path}.title", 0, 1_000)
256
- if "url" in content:
257
- url = expect_string(content["url"], f"{path}.url", 1, 4_096)
258
- parsed = urlsplit(url)
259
- if not parsed.scheme or any(character.isspace() for character in url):
260
- fail(f"{path}.url", "must be an absolute URI")
261
- if "truncated" in content:
262
- expect_bool(content["truncated"], f"{path}.truncated")
263
- return content
264
-
265
-
266
- def validate_provenance(value: object, path: str, content: dict[str, object]) -> None:
267
- provenance = expect_dict(value, path)
268
- exact_keys(
269
- provenance,
270
- path,
271
- {"provider", "contentSha256"},
272
- {"operation", "sourceId", "runId", "policyVersion", "model"},
273
- )
274
- expect_string(provenance["provider"], f"{path}.provider", 1, 100)
275
- for key in ("operation", "policyVersion", "model"):
276
- if key in provenance:
277
- expect_string(provenance[key], f"{path}.{key}", 1, 160)
278
- for key in ("sourceId", "runId"):
279
- if key in provenance:
280
- expect_string(provenance[key], f"{path}.{key}", 1, 300)
281
- digest = expect_digest(provenance["contentSha256"], f"{path}.contentSha256", False)
282
- try:
283
- expected = sha256_hex(canonical_bytes(content))
284
- except ArchiveError as exc:
285
- fail(path, str(exc))
286
- if digest != expected:
287
- fail(path, "content digest mismatch")
288
-
289
-
290
- def validate_record(
291
- value: object,
292
- index: int,
293
- ) -> tuple[dict[str, object], dt.datetime | None, dt.datetime | None]:
294
- path = f"records[{index}]"
295
- record = expect_dict(value, path)
296
- required = {
297
- "id", "digest", "kind", "authorRole", "contentRole", "authorshipConfidence",
298
- "sentStatus", "visibility", "sourceClass", "content", "provenance",
299
- }
300
- exact_keys(record, path, required, {"occurredAt", "observedAt"})
301
- if "occurredAt" not in record and "observedAt" not in record:
302
- fail(path, "must include occurredAt or observedAt")
303
- expect_string(record["id"], f"{path}.id", 1, 200)
304
- digest = expect_digest(record["digest"], f"{path}.digest", True)
305
- expect_string(record["kind"], f"{path}.kind", 1, 100)
306
- expect_enum(record["authorRole"], f"{path}.authorRole", {"subject", "counterpart", "third_party", "mixed", "unknown"})
307
- expect_enum(record["contentRole"], f"{path}.contentRole", {"original", "quoted", "forwarded", "summary", "ai_assisted", "mixed", "unknown"})
308
- expect_enum(record["authorshipConfidence"], f"{path}.authorshipConfidence", {"verified", "strong", "weak", "unknown"})
309
- expect_enum(record["sentStatus"], f"{path}.sentStatus", {"sent", "draft", "received", "published", "unknown"})
310
- expect_enum(record["visibility"], f"{path}.visibility", {"public", "private"})
311
- expect_enum(record["sourceClass"], f"{path}.sourceClass", {
312
- "private_capture", "polished_self_presentation", "observed_behavior", "public_web_evidence",
313
- "third_party_description", "institutional", "metadata",
314
- })
315
- occurred = parse_datetime(record["occurredAt"], f"{path}.occurredAt") if "occurredAt" in record else None
316
- observed = parse_datetime(record["observedAt"], f"{path}.observedAt") if "observedAt" in record else None
317
- content = validate_content(record["content"], f"{path}.content")
318
- validate_provenance(record["provenance"], f"{path}.provenance", content)
319
- without_digest = dict(record)
320
- without_digest.pop("digest")
321
- try:
322
- expected = "sha256:" + sha256_hex(canonical_bytes(without_digest))
323
- except ArchiveError as exc:
324
- fail(path, str(exc))
325
- if digest != expected:
326
- fail(path, "record digest mismatch")
327
- return record, occurred, observed
328
-
329
-
330
- def validate_claims(
331
- value: object,
332
- subject_local_id: str,
333
- record_ids: set[str],
334
- ) -> int:
335
- claims = expect_list(value, "claims")
336
- if len(claims) > 500:
337
- fail("claims", "has too many items")
338
- claim_ids: set[str] = set()
339
- required = {
340
- "id", "text", "recordIds", "status", "claimantRole", "claimKind",
341
- "subjectLocalId", "sensitivity",
342
- }
343
- for index, raw in enumerate(claims):
344
- path = f"claims[{index}]"
345
- claim = expect_dict(raw, path)
346
- exact_keys(claim, path, required)
347
- claim_id = expect_string(claim["id"], f"{path}.id", 1, 200)
348
- if claim_id in claim_ids:
349
- fail(path, "duplicates another claim id")
350
- claim_ids.add(claim_id)
351
- expect_string(claim["text"], f"{path}.text", 1, 4_000)
352
- refs = expect_list(claim["recordIds"], f"{path}.recordIds")
353
- if len(refs) < 1 or len(refs) > 50:
354
- fail(f"{path}.recordIds", "has an invalid item count")
355
- normalized_refs: list[str] = []
356
- for ref_index, ref in enumerate(refs):
357
- normalized_refs.append(expect_string(ref, f"{path}.recordIds[{ref_index}]", 1, 200))
358
- if len(set(normalized_refs)) != len(normalized_refs):
359
- fail(f"{path}.recordIds", "contains duplicates")
360
- if any(ref not in record_ids for ref in normalized_refs):
361
- fail(f"{path}.recordIds", "references an unknown record")
362
- expect_enum(claim["status"], f"{path}.status", {"source_reported", "adapter_structured", "contested"})
363
- expect_enum(claim["claimantRole"], f"{path}.claimantRole", {"subject", "counterpart", "third_party", "institutional", "adapter", "unknown"})
364
- expect_enum(claim["claimKind"], f"{path}.claimKind", {"fact", "stated_belief", "reported_observation", "derived_index"})
365
- if expect_string(claim["subjectLocalId"], f"{path}.subjectLocalId", 1, 200) != subject_local_id:
366
- fail(f"{path}.subjectLocalId", "does not match packet subject")
367
- expect_enum(claim["sensitivity"], f"{path}.sensitivity", {"ordinary", "sensitive_explicit"})
368
- return len(claims)
369
-
370
-
371
- def validate_packet(value: object) -> dict[str, object]:
372
- packet = expect_dict(value, "packet")
373
- required = {
374
- "schemaVersion", "digestCanonicalization", "packetId", "generatedAt", "subject",
375
- "scope", "records", "limitations", "packetDigest",
376
- }
377
- exact_keys(packet, "packet", required, {"claims"})
378
- if packet["schemaVersion"] != "ensoul.source-packet.v1":
379
- fail("schemaVersion", "unsupported schema")
380
- if packet["digestCanonicalization"] != "JCS-RFC8785":
381
- fail("digestCanonicalization", "unsupported digest canonicalization")
382
- expect_string(packet["packetId"], "packetId", 8, 160)
383
- generated_at = parse_datetime(packet["generatedAt"], "generatedAt")
384
- subject = validate_subject(packet["subject"])
385
- scope, limits, as_of, source_cutoff = validate_scope(packet["scope"])
386
- if as_of is not None and as_of > generated_at:
387
- fail("scope.asOf", "must not be later than generatedAt")
388
- if source_cutoff is not None and source_cutoff > generated_at:
389
- fail("scope.sourceCutoff", "must not be later than generatedAt")
390
- if source_cutoff is not None and as_of is not None and source_cutoff > as_of:
391
- fail("scope.sourceCutoff", "must not be later than scope.asOf")
392
- records = expect_list(packet["records"], "records")
393
- if len(records) > 2_000:
394
- fail("records", "has too many items")
395
- record_ids: set[str] = set()
396
- time_values: list[tuple[dt.datetime | None, dt.datetime | None]] = []
397
- for index, raw in enumerate(records):
398
- record, occurred, observed = validate_record(raw, index)
399
- record_id = str(record["id"])
400
- if record_id in record_ids:
401
- fail(f"records[{index}].id", "duplicates another record id")
402
- record_ids.add(record_id)
403
- time_values.append((occurred, observed))
404
- lower, upper = effective_bounds(limits)
405
- for index, (occurred, observed) in enumerate(time_values):
406
- if occurred is not None and observed is not None and occurred > observed:
407
- fail(f"records[{index}]", "occurredAt must not be later than observedAt")
408
- for label, value in (("occurredAt", occurred), ("observedAt", observed)):
409
- if value is None:
410
- continue
411
- if value > generated_at:
412
- fail(f"records[{index}].{label}", "must not be later than generatedAt")
413
- if as_of is not None and value > as_of:
414
- fail(f"records[{index}].{label}", "must not be later than scope.asOf")
415
- if source_cutoff is not None and value > source_cutoff:
416
- fail(f"records[{index}].{label}", "must not be later than scope.sourceCutoff")
417
- effective_time = occurred if occurred is not None else observed
418
- if effective_time is not None and lower is not None and effective_time < lower:
419
- fail(f"records[{index}]", "evidence time is before the declared lower bound")
420
- if effective_time is not None and upper is not None and effective_time >= upper:
421
- fail(f"records[{index}]", "evidence time is at or after the declared upper bound")
422
- limitations = expect_list(packet["limitations"], "limitations")
423
- if len(limitations) < 1 or len(limitations) > 32:
424
- fail("limitations", "has an invalid item count")
425
- for index, limitation in enumerate(limitations):
426
- expect_string(limitation, f"limitations[{index}]", 1, 1_000)
427
- claim_count = validate_claims(packet.get("claims", []), str(subject["localId"]), record_ids)
428
- digest = expect_digest(packet["packetDigest"], "packetDigest", True)
429
- without_digest = dict(packet)
430
- without_digest.pop("packetDigest")
431
- try:
432
- expected = "sha256:" + sha256_hex(canonical_bytes(without_digest))
433
- except ArchiveError as exc:
434
- fail("packet", str(exc))
435
- if digest != expected:
436
- fail("packetDigest", "packet digest mismatch")
437
- visibility_counts = {"private": 0, "public": 0}
438
- for record in records:
439
- visibility_counts[str(record["visibility"])] += 1 # type: ignore[index]
440
- return {
441
- "valid": True,
442
- "schemaVersion": packet["schemaVersion"],
443
- "packetDigest": digest,
444
- "adapter": scope["adapter"],
445
- "payloadSchema": scope["payloadSchema"],
446
- "records": len(records),
447
- "claims": claim_count,
448
- "visibility": visibility_counts,
449
- }
450
-
451
-
452
- def validate_file(path: Path) -> dict[str, object]:
453
- if not path.is_absolute():
454
- fail("input", "path must be absolute")
455
- if path.is_symlink() or not path.is_file():
456
- fail("input", "must be a regular non-symlink file")
457
- size = os.stat(path, follow_symlinks=False).st_size
458
- if size > MAX_PACKET_BYTES:
459
- fail("input", "packet exceeds the size limit")
460
- return validate_packet(strict_json_loads(path.read_bytes()))
461
-
462
-
463
- def main(argv: list[str] | None = None) -> int:
464
- parser = argparse.ArgumentParser(description=__doc__)
465
- parser.add_argument("packet", type=Path, help="absolute path to one Ensoul packet")
466
- args = parser.parse_args(sys.argv[1:] if argv is None else argv)
467
- try:
468
- receipt = validate_file(args.packet)
469
- print(json.dumps(receipt, sort_keys=True, separators=(",", ":")))
470
- return 0
471
- except (PacketValidationError, OSError) as exc:
472
- print(f"invalid Ensoul source packet: {exc}", file=sys.stderr)
473
- return 2
474
-
475
-
476
- if __name__ == "__main__":
477
- raise SystemExit(main())