ref-id 0.7.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ref_id/__init__.py +69 -0
- ref_id/build.py +146 -0
- ref_id/canonical.py +171 -0
- ref_id/digest.py +45 -0
- ref_id/encoding.py +78 -0
- ref_id/envelope.py +64 -0
- ref_id/errors.py +65 -0
- ref_id/grammar.py +92 -0
- ref_id/parse.py +236 -0
- ref_id/py.typed +0 -0
- ref_id/relations.py +464 -0
- ref_id/serialise.py +109 -0
- ref_id/spec/ref-id.json +5258 -0
- ref_id/spec/ref-id.json.sha256 +1 -0
- ref_id/spec.py +192 -0
- ref_id/types.py +264 -0
- ref_id/validators.py +258 -0
- ref_id-0.7.0.dist-info/METADATA +70 -0
- ref_id-0.7.0.dist-info/RECORD +21 -0
- ref_id-0.7.0.dist-info/WHEEL +4 -0
- ref_id-0.7.0.dist-info/licenses/LICENSE +202 -0
ref_id/__init__.py
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""The `ref:` identifier scheme — a Python port held to `spec/ref-id.json`'s conformance vectors.
|
|
3
|
+
|
|
4
|
+
Track 2 lands the embedded specification, its integrity check, the canonical serialisation and the set
|
|
5
|
+
digest. Parse, serialise, build and the envelope are Track 3; relations are Track 4.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from .build import build
|
|
10
|
+
from .canonical import canonicalise
|
|
11
|
+
from .digest import digest
|
|
12
|
+
from .envelope import validate_envelope
|
|
13
|
+
from .errors import (
|
|
14
|
+
BuildError,
|
|
15
|
+
DigestError,
|
|
16
|
+
RefIdError,
|
|
17
|
+
SerialiseError,
|
|
18
|
+
SpecIntegrityError,
|
|
19
|
+
SpecVersionError,
|
|
20
|
+
)
|
|
21
|
+
from .parse import parse
|
|
22
|
+
from .relations import covers, relate, same_identifier, same_package, verdict
|
|
23
|
+
from .serialise import canonical_identifier, serialise
|
|
24
|
+
from .spec import Spec, embedded_spec_text, load_spec, load_spec_from
|
|
25
|
+
from .types import (
|
|
26
|
+
BuildParts,
|
|
27
|
+
EnvelopeResult,
|
|
28
|
+
Fragment,
|
|
29
|
+
NestedValue,
|
|
30
|
+
ParseResult,
|
|
31
|
+
QualifierRelation,
|
|
32
|
+
RelateResult,
|
|
33
|
+
VerdictDecidedBy,
|
|
34
|
+
VerdictResult,
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
__all__ = [
|
|
38
|
+
"BuildError",
|
|
39
|
+
"BuildParts",
|
|
40
|
+
"DigestError",
|
|
41
|
+
"EnvelopeResult",
|
|
42
|
+
"Fragment",
|
|
43
|
+
"NestedValue",
|
|
44
|
+
"ParseResult",
|
|
45
|
+
"QualifierRelation",
|
|
46
|
+
"RefIdError",
|
|
47
|
+
"RelateResult",
|
|
48
|
+
"SerialiseError",
|
|
49
|
+
"Spec",
|
|
50
|
+
"SpecIntegrityError",
|
|
51
|
+
"SpecVersionError",
|
|
52
|
+
"VerdictDecidedBy",
|
|
53
|
+
"VerdictResult",
|
|
54
|
+
"build",
|
|
55
|
+
"canonical_identifier",
|
|
56
|
+
"canonicalise",
|
|
57
|
+
"covers",
|
|
58
|
+
"digest",
|
|
59
|
+
"embedded_spec_text",
|
|
60
|
+
"load_spec",
|
|
61
|
+
"load_spec_from",
|
|
62
|
+
"parse",
|
|
63
|
+
"relate",
|
|
64
|
+
"same_identifier",
|
|
65
|
+
"same_package",
|
|
66
|
+
"serialise",
|
|
67
|
+
"validate_envelope",
|
|
68
|
+
"verdict",
|
|
69
|
+
]
|
ref_id/build.py
ADDED
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""Assembles a `ref:` identifier string from its parts. A part the grammar cannot carry is refused with
|
|
3
|
+
`BuildError` naming it — the builder never emits a string that means something else, which is proven at
|
|
4
|
+
the end by parsing what was built and comparing it with what was asked.
|
|
5
|
+
|
|
6
|
+
Ported from `crates/ref-id/src/build.rs`.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from .encoding import contains_any, encode, table_for
|
|
13
|
+
from .errors import BuildError
|
|
14
|
+
from .grammar import (
|
|
15
|
+
FIELD,
|
|
16
|
+
FRAGMENT_INTRODUCER,
|
|
17
|
+
LINE_BREAKS,
|
|
18
|
+
PAIR,
|
|
19
|
+
scheme_prefix,
|
|
20
|
+
)
|
|
21
|
+
from .parse import parse
|
|
22
|
+
from .spec import Spec, load_spec
|
|
23
|
+
from .types import BuildParts, Fragment, NestedValue
|
|
24
|
+
from .validators import folds_type
|
|
25
|
+
|
|
26
|
+
__all__ = ["build"]
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _nesting_form(spec: Spec, key: str) -> dict[str, Any] | None:
|
|
30
|
+
declared = spec.get_object("qualifiers")
|
|
31
|
+
entry = declared.get(key) if declared is not None else None
|
|
32
|
+
if not isinstance(entry, dict):
|
|
33
|
+
return None
|
|
34
|
+
forms = spec.get_object("forms") or {}
|
|
35
|
+
for name in entry.get("forms", []):
|
|
36
|
+
if not isinstance(name, str):
|
|
37
|
+
continue
|
|
38
|
+
form = forms.get(name)
|
|
39
|
+
if isinstance(form, dict) and form.get("nested") is True:
|
|
40
|
+
return form
|
|
41
|
+
return None
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _refuse(spec: Spec, part: str, why: str) -> BuildError:
|
|
45
|
+
return _refuse_at(spec.part(part), why)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _refuse_at(part: str, why: str) -> BuildError:
|
|
49
|
+
"""The message names the refused part, so a caller reads what to change instead of guessing and
|
|
50
|
+
retrying. A part `parse` already reported is used verbatim: it may be a key the identifier carries
|
|
51
|
+
and the spec never declared."""
|
|
52
|
+
return BuildError(part, f"cannot build: {why} — refused at the {part}")
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def build(parts: BuildParts, /) -> str:
|
|
56
|
+
"""Builds a `ref:` identifier string."""
|
|
57
|
+
spec = load_spec()
|
|
58
|
+
grammar = spec.grammar()
|
|
59
|
+
state_separator = spec.get_str("grammar", "state", "separator")
|
|
60
|
+
fragment_separator = spec.get_str("grammar", "fragment", "separator")
|
|
61
|
+
separators = state_separator + FRAGMENT_INTRODUCER + "".join(LINE_BREAKS)
|
|
62
|
+
fragment_reserved = fragment_separator + "".join(LINE_BREAKS)
|
|
63
|
+
|
|
64
|
+
locator = parts.locator
|
|
65
|
+
fold = f"{parts.type}{FIELD}"
|
|
66
|
+
if folds_type(spec, parts.type) and locator.startswith(fold):
|
|
67
|
+
locator = locator[len(fold) :]
|
|
68
|
+
|
|
69
|
+
if not locator or contains_any(locator, separators):
|
|
70
|
+
raise _refuse(spec, "locator", "a locator is handed to its validator verbatim and cannot carry a reserved character")
|
|
71
|
+
|
|
72
|
+
out = [f"{scheme_prefix(spec)}{parts.type}{FIELD}{locator}"]
|
|
73
|
+
|
|
74
|
+
if parts.qualifiers:
|
|
75
|
+
rendered = []
|
|
76
|
+
for pair in parts.qualifiers:
|
|
77
|
+
# `parts.qualifiers` is declared as a tuple of `(key, value)` pairs; a caller who hands a
|
|
78
|
+
# `dict` instead bypasses the type checker (as `BuildParts.from_json`'s own malformed-shape
|
|
79
|
+
# guard already does for the JSON entry point) and would otherwise unpack each dict *key*
|
|
80
|
+
# string into `key, value` — silently wrong for a two-character key, a bare `ValueError` for
|
|
81
|
+
# any other length. Checked and refused here the same way `from_json` refuses its own shape.
|
|
82
|
+
if not isinstance(pair, (tuple, list)) or len(pair) != 2 or not isinstance(pair[0], str):
|
|
83
|
+
raise _refuse(spec, "state", "a qualifier is a (key, value) pair with a string key")
|
|
84
|
+
key, value = pair
|
|
85
|
+
if isinstance(value, NestedValue):
|
|
86
|
+
if not isinstance(value.nested, str):
|
|
87
|
+
raise _refuse(spec, key, "a qualifier value is a string or NestedValue holding a string")
|
|
88
|
+
form = _nesting_form(spec, key)
|
|
89
|
+
if form is None:
|
|
90
|
+
raise _refuse(spec, key, "this qualifier declares no nesting form")
|
|
91
|
+
encoded = encode(value.nested, table_for(spec, form))
|
|
92
|
+
elif isinstance(value, str):
|
|
93
|
+
if value.startswith(scheme_prefix(spec)) or contains_any(value, separators):
|
|
94
|
+
raise _refuse(spec, key, "a nested identifier is passed as Nested, never as a plain string")
|
|
95
|
+
encoded = value
|
|
96
|
+
else:
|
|
97
|
+
raise _refuse(spec, key, "a qualifier value is a string or NestedValue holding a string")
|
|
98
|
+
rendering = f"{key}{PAIR}{encoded}"
|
|
99
|
+
if grammar.state_pair.fullmatch(rendering) is None:
|
|
100
|
+
raise _refuse(spec, key, "the key does not fit the pair grammar")
|
|
101
|
+
rendered.append(rendering)
|
|
102
|
+
out.append(state_separator)
|
|
103
|
+
out.append(state_separator.join(rendered))
|
|
104
|
+
|
|
105
|
+
wanted_path: str | None = None
|
|
106
|
+
wanted_refinements: tuple[tuple[str, str], ...] = ()
|
|
107
|
+
if parts.fragment is not None:
|
|
108
|
+
if isinstance(parts.fragment, str):
|
|
109
|
+
path = parts.fragment
|
|
110
|
+
elif isinstance(parts.fragment, Fragment):
|
|
111
|
+
path = parts.fragment.path
|
|
112
|
+
wanted_refinements = parts.fragment.refinements
|
|
113
|
+
else:
|
|
114
|
+
raise _refuse(spec, "fragment", "a fragment is a string or Fragment")
|
|
115
|
+
if not path or contains_any(path, fragment_reserved):
|
|
116
|
+
raise _refuse(spec, "fragment", "a declared-name path cannot be empty or carry the refinement separator")
|
|
117
|
+
for key, value in wanted_refinements:
|
|
118
|
+
if contains_any(value, fragment_reserved):
|
|
119
|
+
raise _refuse(spec, key, "a refinement value cannot carry the separator")
|
|
120
|
+
out.append(FRAGMENT_INTRODUCER)
|
|
121
|
+
out.append(path)
|
|
122
|
+
if wanted_refinements:
|
|
123
|
+
out.append(fragment_separator)
|
|
124
|
+
out.append(fragment_separator.join(f"{key}{PAIR}{value}" for key, value in wanted_refinements))
|
|
125
|
+
wanted_path = path
|
|
126
|
+
|
|
127
|
+
assembled = "".join(out)
|
|
128
|
+
|
|
129
|
+
# The last word is the grammar's: what was built must decompose to exactly what was asked.
|
|
130
|
+
if grammar.top.fullmatch(assembled) is None:
|
|
131
|
+
raise _refuse(spec, "grammar", "the assembled string does not match the grammar")
|
|
132
|
+
check = parse(assembled)
|
|
133
|
+
if check.status == spec.status("malformed"):
|
|
134
|
+
refused = check.part if check.part is not None else spec.part("grammar")
|
|
135
|
+
raise _refuse_at(refused, "the assembled string is malformed")
|
|
136
|
+
if check.explicit_version or check.type != parts.type:
|
|
137
|
+
raise _refuse(spec, "type", "the type re-split into other parts")
|
|
138
|
+
if check.locator != locator:
|
|
139
|
+
raise _refuse(spec, "locator", "the locator re-split into other parts")
|
|
140
|
+
if len(check.qualifiers) != len(parts.qualifiers):
|
|
141
|
+
raise _refuse(spec, "state", "a qualifier re-split into other parts")
|
|
142
|
+
check_path = check.fragment.path if check.fragment is not None else None
|
|
143
|
+
check_refinements = len(check.fragment.refinements) if check.fragment is not None else 0
|
|
144
|
+
if check_path != wanted_path or check_refinements != len(wanted_refinements):
|
|
145
|
+
raise _refuse(spec, "fragment", "the fragment re-split into other parts")
|
|
146
|
+
return assembled
|
ref_id/canonical.py
ADDED
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""The canonical serialisation `/canonicalisation` fixes: object keys sorted by UTF-16 code unit, no
|
|
3
|
+
whitespace outside strings, strings escaped exactly as ECMAScript `JSON.stringify` does, integers only.
|
|
4
|
+
|
|
5
|
+
Ported from `crates/ref-id/src/canonical.rs`. `json.dumps` is not used for string escaping: measured
|
|
6
|
+
(2026-09-27), Python's own encoder writes a lone surrogate through unescaped, where `JSON.stringify`
|
|
7
|
+
escapes it as `\\uXXXX` — the two implementations would then digest different bytes for the same
|
|
8
|
+
member.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
|
|
14
|
+
from .errors import SpecIntegrityError
|
|
15
|
+
|
|
16
|
+
__all__ = ["canonicalise"]
|
|
17
|
+
|
|
18
|
+
_maximum_cache: int | None = None
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _maximum() -> int:
|
|
22
|
+
"""`version.maximum`, read from the embedded specification rather than written as a literal here.
|
|
23
|
+
Deferred import: `spec.py` imports this module at its own top level to build `canonicalise`, so the
|
|
24
|
+
reverse read happens only inside this function body, by which time both modules are fully loaded —
|
|
25
|
+
and it reads the raw bytes rather than the validated `Spec`, because this is itself the function that
|
|
26
|
+
computes the digest `Spec` is validated against."""
|
|
27
|
+
global _maximum_cache
|
|
28
|
+
if _maximum_cache is None:
|
|
29
|
+
from .spec import embedded_spec_text
|
|
30
|
+
|
|
31
|
+
json_text, _sidecar_text = embedded_spec_text()
|
|
32
|
+
document = json.loads(json_text)
|
|
33
|
+
value = document.get("version", {}).get("maximum") if isinstance(document, dict) else None
|
|
34
|
+
if not isinstance(value, int) or isinstance(value, bool):
|
|
35
|
+
raise SpecIntegrityError("spec/ref-id.json does not declare version.maximum as an integer")
|
|
36
|
+
_maximum_cache = value
|
|
37
|
+
return _maximum_cache
|
|
38
|
+
|
|
39
|
+
_ESCAPES = {
|
|
40
|
+
'"': '\\"',
|
|
41
|
+
"\\": "\\\\",
|
|
42
|
+
"\b": "\\b",
|
|
43
|
+
"\f": "\\f",
|
|
44
|
+
"\n": "\\n",
|
|
45
|
+
"\r": "\\r",
|
|
46
|
+
"\t": "\\t",
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _escape(string: str) -> str:
|
|
51
|
+
out: list[str] = ['"']
|
|
52
|
+
index = 0
|
|
53
|
+
length = len(string)
|
|
54
|
+
while index < length:
|
|
55
|
+
char = string[index]
|
|
56
|
+
code = ord(char)
|
|
57
|
+
# A Python `str` may hold an unpaired surrogate half where JavaScript would already have joined
|
|
58
|
+
# it into one UTF-16 code unit pair naming a single character. Joining a high/low pair back into
|
|
59
|
+
# its codepoint before deciding how to write it keeps this port's output identical to
|
|
60
|
+
# `JSON.stringify`'s for a string built the way JavaScript itself would see it — two `str` halves
|
|
61
|
+
# from splitting an astral character must canonicalise the same as that one character.
|
|
62
|
+
if 0xD800 <= code <= 0xDBFF and index + 1 < length:
|
|
63
|
+
next_code = ord(string[index + 1])
|
|
64
|
+
if 0xDC00 <= next_code <= 0xDFFF:
|
|
65
|
+
combined = 0x10000 + (code - 0xD800) * 0x400 + (next_code - 0xDC00)
|
|
66
|
+
out.append(chr(combined))
|
|
67
|
+
index += 2
|
|
68
|
+
continue
|
|
69
|
+
if char in _ESCAPES:
|
|
70
|
+
out.append(_ESCAPES[char])
|
|
71
|
+
elif code < 0x20 or 0xD800 <= code <= 0xDFFF:
|
|
72
|
+
# A lone surrogate cannot be re-encoded as UTF-8 unescaped; `JSON.stringify` escapes it
|
|
73
|
+
# exactly like a control character, and this port must match that byte for byte.
|
|
74
|
+
out.append(f"\\u{code:04x}")
|
|
75
|
+
else:
|
|
76
|
+
out.append(char)
|
|
77
|
+
index += 1
|
|
78
|
+
out.append('"')
|
|
79
|
+
return "".join(out)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _sort_key(key: str) -> bytes:
|
|
83
|
+
# UTF-16 code unit order, per `/canonicalisation`. `surrogatepass` lets a key that is itself a lone
|
|
84
|
+
# surrogate (or carries one) still sort, rather than raising where the specification does not.
|
|
85
|
+
return key.encode("utf-16-be", "surrogatepass")
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
# A depth this deep has no legitimate use in a specification document (the real one nests a handful of
|
|
89
|
+
# levels); it exists so a pathologically deep or cyclic input is refused with `SpecIntegrityError` well
|
|
90
|
+
# before it could exhaust CPython's own call stack (`RecursionError`) — Security review finding 5.
|
|
91
|
+
_MAX_DEPTH = 500
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _describe_int(value: int) -> str:
|
|
95
|
+
"""A safe textual magnitude for an error message — `str()`/`repr()` on an integer long enough hits
|
|
96
|
+
CPython's own int-conversion digit-count guard, which would turn *reporting* the refusal into the
|
|
97
|
+
same crash this function exists to avoid."""
|
|
98
|
+
try:
|
|
99
|
+
return repr(value)
|
|
100
|
+
except ValueError:
|
|
101
|
+
sign = "-" if value < 0 else ""
|
|
102
|
+
return f"{sign}<integer of magnitude 2**{value.bit_length()}>"
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _write(value: object, out: list[str], maximum: int, depth: int, seen: set[int]) -> None:
|
|
106
|
+
if depth > _MAX_DEPTH:
|
|
107
|
+
raise SpecIntegrityError(f"canonicalisation nests more than {_MAX_DEPTH} levels deep")
|
|
108
|
+
if value is None:
|
|
109
|
+
out.append("null")
|
|
110
|
+
elif isinstance(value, bool):
|
|
111
|
+
out.append("true" if value else "false")
|
|
112
|
+
elif isinstance(value, int):
|
|
113
|
+
if abs(value) > maximum:
|
|
114
|
+
raise SpecIntegrityError(f"canonicalisation covers magnitudes up to {maximum}, got {_describe_int(value)}")
|
|
115
|
+
out.append(str(value))
|
|
116
|
+
elif isinstance(value, float):
|
|
117
|
+
# A number is an integer whose magnitude is at most `version.maximum`, written as that integer;
|
|
118
|
+
# an integral value written with a fraction or an exponent (`json.loads` turns both into a
|
|
119
|
+
# `float`, e.g. `1.0` or `1e2`) is that integer; any other number is refused (`/canonicalisation`).
|
|
120
|
+
if not value.is_integer() or abs(value) > maximum:
|
|
121
|
+
raise SpecIntegrityError(f"canonicalisation covers integers only, got {value!r}")
|
|
122
|
+
out.append(str(int(value)))
|
|
123
|
+
elif isinstance(value, str):
|
|
124
|
+
out.append(_escape(value))
|
|
125
|
+
elif isinstance(value, (list, tuple)):
|
|
126
|
+
# `id(value)` marks a container as "on the path from the root to here" for the length of its own
|
|
127
|
+
# recursion (added on entry, discarded on exit) — a reference cycle (a container that reaches
|
|
128
|
+
# itself through its own descendants) is refused instead of recursing forever, while the same
|
|
129
|
+
# object appearing twice in separate, non-nested branches (aliasing, not a cycle) still
|
|
130
|
+
# canonicalises normally.
|
|
131
|
+
marker = id(value)
|
|
132
|
+
if marker in seen:
|
|
133
|
+
raise SpecIntegrityError("canonicalisation cannot serialise a cyclic structure")
|
|
134
|
+
seen.add(marker)
|
|
135
|
+
try:
|
|
136
|
+
out.append("[")
|
|
137
|
+
for i, item in enumerate(value):
|
|
138
|
+
if i > 0:
|
|
139
|
+
out.append(",")
|
|
140
|
+
_write(item, out, maximum, depth + 1, seen)
|
|
141
|
+
out.append("]")
|
|
142
|
+
finally:
|
|
143
|
+
seen.discard(marker)
|
|
144
|
+
elif isinstance(value, dict):
|
|
145
|
+
marker = id(value)
|
|
146
|
+
if marker in seen:
|
|
147
|
+
raise SpecIntegrityError("canonicalisation cannot serialise a cyclic structure")
|
|
148
|
+
seen.add(marker)
|
|
149
|
+
try:
|
|
150
|
+
if any(not isinstance(key, str) for key in value):
|
|
151
|
+
raise SpecIntegrityError("canonicalisation covers string-keyed objects only")
|
|
152
|
+
keys = sorted(value.keys(), key=_sort_key)
|
|
153
|
+
out.append("{")
|
|
154
|
+
for i, key in enumerate(keys):
|
|
155
|
+
if i > 0:
|
|
156
|
+
out.append(",")
|
|
157
|
+
out.append(_escape(key))
|
|
158
|
+
out.append(":")
|
|
159
|
+
_write(value[key], out, maximum, depth + 1, seen)
|
|
160
|
+
out.append("}")
|
|
161
|
+
finally:
|
|
162
|
+
seen.discard(marker)
|
|
163
|
+
else:
|
|
164
|
+
raise SpecIntegrityError(f"canonicalisation cannot serialise a value of type {type(value).__name__}")
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def canonicalise(value: object, /) -> str:
|
|
168
|
+
"""The canonical serialisation the specification digest is computed over."""
|
|
169
|
+
out: list[str] = []
|
|
170
|
+
_write(value, out, _maximum(), 0, set())
|
|
171
|
+
return "".join(out)
|
ref_id/digest.py
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""sha256 over the UTF-8 bytes of the joined identifier strings, per `spec.digest`: declared order, no
|
|
3
|
+
deduplication, and a refusal for a member that carries the join character.
|
|
4
|
+
|
|
5
|
+
Ported line for line from `crates/ref-id/src/digest.rs`.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import hashlib
|
|
10
|
+
from collections.abc import Sequence
|
|
11
|
+
|
|
12
|
+
from .errors import DigestError, SpecVersionError
|
|
13
|
+
from .grammar import FIELD
|
|
14
|
+
from .spec import load_spec
|
|
15
|
+
|
|
16
|
+
__all__ = ["digest"]
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def digest(members: Sequence[str], /) -> str:
|
|
20
|
+
"""Digests an ordered, non-deduplicated sequence of identifier strings."""
|
|
21
|
+
spec = load_spec()
|
|
22
|
+
if not isinstance(members, (list, tuple)):
|
|
23
|
+
# A bare `str` satisfies `Sequence[str]` at the type level (of its own characters) but is not a
|
|
24
|
+
# sequence of members, and neither is a generator or a set — the TypeScript reference refuses
|
|
25
|
+
# anything that is not an array the same way (`packages/ref-id/src/digest.ts:15`).
|
|
26
|
+
raise DigestError(spec.part("member"), "members must be a list or tuple of strings")
|
|
27
|
+
# One snapshot, taken only after the shape refusal above, so a lazily-consumed iterable (were one
|
|
28
|
+
# ever accepted) cannot be re-read differently between validation and hashing.
|
|
29
|
+
snapshot = list(members)
|
|
30
|
+
join = spec.get_str("digest", "join")
|
|
31
|
+
for member in snapshot:
|
|
32
|
+
if not isinstance(member, str):
|
|
33
|
+
raise DigestError(spec.part("member"), "a member must be a string")
|
|
34
|
+
try:
|
|
35
|
+
member.encode("utf-8")
|
|
36
|
+
except UnicodeEncodeError as error:
|
|
37
|
+
raise DigestError(spec.part("member"), "a member must encode as UTF-8") from error
|
|
38
|
+
if join in member:
|
|
39
|
+
raise DigestError(spec.part("member"), "a member must be a string that does not carry the join character")
|
|
40
|
+
algorithm = spec.get_str("digest", "algorithm")
|
|
41
|
+
if algorithm != "sha256":
|
|
42
|
+
raise SpecVersionError(f"spec.digest.algorithm declares {algorithm}; this package implements sha256 only")
|
|
43
|
+
joined = join.join(snapshot)
|
|
44
|
+
hashed = hashlib.sha256(joined.encode("utf-8")).hexdigest()
|
|
45
|
+
return f"{algorithm}{FIELD}{hashed}"
|
ref_id/encoding.py
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""Percent-encode/decode helpers driven by an `encoding.<name>.table` from the specification — never a
|
|
3
|
+
hardcoded character list. Which table applies to a form is the form's own `encoding` field.
|
|
4
|
+
|
|
5
|
+
Ported from `crates/ref-id/src/encoding.rs`.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
from .spec import Spec
|
|
12
|
+
|
|
13
|
+
__all__: list[str] = []
|
|
14
|
+
|
|
15
|
+
Table = list[tuple[str, str]]
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def table_for(spec: Spec, form: dict[str, Any]) -> Table:
|
|
19
|
+
"""The encoding table a form declares, or an empty one when the form declares no encoding."""
|
|
20
|
+
name = form.get("encoding")
|
|
21
|
+
if not isinstance(name, str):
|
|
22
|
+
return []
|
|
23
|
+
return spec.get_table("encoding", name, "table")
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def contains_any(raw: str, characters: str) -> bool:
|
|
27
|
+
return any(char in characters for char in raw)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def encode(raw: str, table: Table) -> str:
|
|
31
|
+
"""One left-to-right pass over the source characters; a produced percent-form is never re-scanned."""
|
|
32
|
+
if not table:
|
|
33
|
+
return raw
|
|
34
|
+
out: list[str] = []
|
|
35
|
+
for char in raw:
|
|
36
|
+
form = next((value for key, value in table if len(key) == 1 and key == char), None)
|
|
37
|
+
out.append(form if form is not None else char)
|
|
38
|
+
return "".join(out)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def decode(encoded: str, table: Table) -> str:
|
|
42
|
+
"""One left-to-right pass over the encoded text; decoding `%2523` yields `%23`, never `#`.
|
|
43
|
+
|
|
44
|
+
Scans by index rather than slicing a fresh `rest` string every step: `rest = rest[1:]` (and the
|
|
45
|
+
matching `rest.startswith(form)`) copies the remaining text on every character, which is quadratic in
|
|
46
|
+
the input length. `str.startswith(form, i)` matches at an offset without copying anything."""
|
|
47
|
+
if not table:
|
|
48
|
+
return encoded
|
|
49
|
+
out: list[str] = []
|
|
50
|
+
length = len(encoded)
|
|
51
|
+
index = 0
|
|
52
|
+
while index < length:
|
|
53
|
+
match = next(((character, form) for character, form in table if encoded.startswith(form, index)), None)
|
|
54
|
+
if match is not None:
|
|
55
|
+
character, form = match
|
|
56
|
+
out.append(character)
|
|
57
|
+
index += len(form)
|
|
58
|
+
else:
|
|
59
|
+
out.append(encoded[index])
|
|
60
|
+
index += 1
|
|
61
|
+
return "".join(out)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def strictly_encoded(value: str, table: Table) -> bool:
|
|
65
|
+
"""True when every `%` in the value begins one of the table's percent-forms — the strict nested
|
|
66
|
+
encoding.
|
|
67
|
+
|
|
68
|
+
`value[at:]` builds a fresh substring at every `%` found, which is quadratic when `%` recurs densely
|
|
69
|
+
(`"%25" * n`, say); `value.startswith(form, at)` checks the same forms at that offset without copying
|
|
70
|
+
the tail."""
|
|
71
|
+
index = 0
|
|
72
|
+
while True:
|
|
73
|
+
at = value.find("%", index)
|
|
74
|
+
if at < 0:
|
|
75
|
+
return True
|
|
76
|
+
if not any(value.startswith(form, at) for _character, form in table):
|
|
77
|
+
return False
|
|
78
|
+
index = at + 1
|
ref_id/envelope.py
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""The envelope invariant from `spec.envelope`: an identifier carrying a digest is admissible only with
|
|
3
|
+
an object whose sets entry for that qualifier recomputes to it, served under its own id.
|
|
4
|
+
|
|
5
|
+
Ported from `crates/ref-id/src/envelope.rs`.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from .digest import digest
|
|
10
|
+
from .errors import DigestError, SpecVersionError
|
|
11
|
+
from .parse import parse
|
|
12
|
+
from .spec import load_spec
|
|
13
|
+
from .types import EnvelopeResult
|
|
14
|
+
|
|
15
|
+
__all__ = ["validate_envelope"]
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _refused(reason: str) -> EnvelopeResult:
|
|
19
|
+
return EnvelopeResult(admissible=False, reason=reason)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def validate_envelope(requested_id: str, envelope: object) -> EnvelopeResult:
|
|
23
|
+
"""Refuses an envelope whose self-reference or recomputed digests do not hold."""
|
|
24
|
+
spec = load_spec()
|
|
25
|
+
self_reference = spec.get_str("envelope", "selfReference")
|
|
26
|
+
sets_field = spec.get_str("envelope", "setsField")
|
|
27
|
+
|
|
28
|
+
if not isinstance(envelope, dict):
|
|
29
|
+
return _refused("envelope is not an object")
|
|
30
|
+
if envelope.get(self_reference) != requested_id:
|
|
31
|
+
return _refused(f"envelope.{self_reference} does not match the requested identifier")
|
|
32
|
+
|
|
33
|
+
parsed = parse(requested_id)
|
|
34
|
+
if parsed.status in (spec.status("malformed"), spec.status("unsupported")):
|
|
35
|
+
return _refused(f"the requested identifier is {parsed.status}")
|
|
36
|
+
|
|
37
|
+
grammar = spec.grammar()
|
|
38
|
+
digest_patterns: list[str] = []
|
|
39
|
+
for name, form in (spec.get_object("forms") or {}).items():
|
|
40
|
+
if not isinstance(form, dict) or form.get("digest") is not True:
|
|
41
|
+
continue
|
|
42
|
+
pattern = form.get("pattern")
|
|
43
|
+
if not isinstance(pattern, str):
|
|
44
|
+
raise SpecVersionError(f"spec.forms.{name} declares digest: true without a pattern; this package cannot recognise it")
|
|
45
|
+
digest_patterns.append(pattern)
|
|
46
|
+
|
|
47
|
+
sets = envelope.get(sets_field)
|
|
48
|
+
sets = sets if isinstance(sets, dict) else None
|
|
49
|
+
|
|
50
|
+
for key, value in parsed.qualifiers:
|
|
51
|
+
is_digest = any(grammar.matches(spec, pattern, value) for pattern in digest_patterns)
|
|
52
|
+
if not is_digest:
|
|
53
|
+
continue
|
|
54
|
+
members = sets.get(key) if sets is not None else None
|
|
55
|
+
if not isinstance(members, list) or not all(isinstance(item, str) for item in members):
|
|
56
|
+
return _refused(f"{sets_field}.{key} is missing or is not an array of strings")
|
|
57
|
+
try:
|
|
58
|
+
recomputed = digest(members)
|
|
59
|
+
except DigestError as error:
|
|
60
|
+
return _refused(f"{sets_field}.{key} {error.message}")
|
|
61
|
+
if recomputed != value:
|
|
62
|
+
return _refused(f"{sets_field}.{key} does not recompute to the declared digest")
|
|
63
|
+
|
|
64
|
+
return EnvelopeResult(admissible=True)
|
ref_id/errors.py
ADDED
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""The error hierarchy every other module raises through.
|
|
3
|
+
|
|
4
|
+
`RefIdError` is the base every kind subclasses; `BuildError`, `DigestError` and `SerialiseError` carry
|
|
5
|
+
the refused `part`, `SpecIntegrityError` and `SpecVersionError` do not. `str(error)` follows the crate's
|
|
6
|
+
`Display` (`crates/ref-id/src/types.rs:142`): ``"<Kind>Error at <part>: <message>"`` for the three with a
|
|
7
|
+
part, ``"<Kind>Error: <message>"`` for the other two.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
__all__ = [
|
|
12
|
+
"BuildError",
|
|
13
|
+
"DigestError",
|
|
14
|
+
"RefIdError",
|
|
15
|
+
"SerialiseError",
|
|
16
|
+
"SpecIntegrityError",
|
|
17
|
+
"SpecVersionError",
|
|
18
|
+
]
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class RefIdError(ValueError):
|
|
22
|
+
"""The base of every error this package raises. `parse` raises none for an identifier problem."""
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class BuildError(RefIdError):
|
|
26
|
+
"""`build` was given a part the grammar cannot carry."""
|
|
27
|
+
|
|
28
|
+
def __init__(self, part: str, message: str) -> None:
|
|
29
|
+
super().__init__(f"BuildError at {part}: {message}")
|
|
30
|
+
self.part = part
|
|
31
|
+
self.message = message
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class DigestError(RefIdError):
|
|
35
|
+
"""`digest` was given a member that is not one identifier string."""
|
|
36
|
+
|
|
37
|
+
def __init__(self, part: str, message: str) -> None:
|
|
38
|
+
super().__init__(f"DigestError at {part}: {message}")
|
|
39
|
+
self.part = part
|
|
40
|
+
self.message = message
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class SerialiseError(RefIdError):
|
|
44
|
+
"""`serialise` was given a result that has no faithful string form."""
|
|
45
|
+
|
|
46
|
+
def __init__(self, part: str, message: str) -> None:
|
|
47
|
+
super().__init__(f"SerialiseError at {part}: {message}")
|
|
48
|
+
self.part = part
|
|
49
|
+
self.message = message
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class SpecIntegrityError(RefIdError):
|
|
53
|
+
"""The embedded specification does not match its sidecar digest."""
|
|
54
|
+
|
|
55
|
+
def __init__(self, message: str) -> None:
|
|
56
|
+
super().__init__(f"SpecIntegrityError: {message}")
|
|
57
|
+
self.message = message
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class SpecVersionError(RefIdError):
|
|
61
|
+
"""The embedded specification declares a version or vocabulary this package does not support."""
|
|
62
|
+
|
|
63
|
+
def __init__(self, message: str) -> None:
|
|
64
|
+
super().__init__(f"SpecVersionError: {message}")
|
|
65
|
+
self.message = message
|