ref-id 0.7.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
ref_id/relations.py ADDED
@@ -0,0 +1,464 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """The three questions equality cannot answer. `same_identifier` says whether two strings name one thing,
3
+ which is deliberately strict. `same_package` says whether two identifiers name the same released thing at
4
+ whatever version each declares; `covers` says whether a partial identifier stands for a whole family of
5
+ complete ones. `relate` reports the relation in every dimension instead of reducing it to one boolean;
6
+ `covers`, `coveredBy` and `same_package` are stated as reductions of that same result, so they are
7
+ implemented as reductions here too, rather than as three computations that could disagree.
8
+
9
+ Ported from `crates/ref-id/src/relations.rs` — read that file's comments for the full rationale; this
10
+ restates none of the spec's own tables.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ from .errors import RefIdError
15
+ from .parse import parse
16
+ from .serialise import canonical_identifier
17
+ from .spec import Spec, load_spec
18
+ from .types import (
19
+ Pair,
20
+ ParseResult,
21
+ QualifierRelation,
22
+ RelateResult,
23
+ Relation,
24
+ VerdictContent,
25
+ VerdictDecidedBy,
26
+ VerdictIdentity,
27
+ VerdictResult,
28
+ )
29
+
30
+ __all__ = ["covers", "relate", "same_identifier", "same_package", "verdict"]
31
+
32
+ # The specification's `IdentifierOrParsed`: a bare string, read with `parse`, or an already-parsed
33
+ # `ParseResult`, used as is. A pair may mix them freely.
34
+ IdentifierOrParsed = str | ParseResult
35
+
36
+ # The fixed dimensions `verdict`'s step one and step five walk, in the order `decidedBy` fixes them
37
+ # (`comparison.verdict.result.decidedBy`).
38
+ _FIXED_DIMENSIONS: tuple[str, ...] = ("type", "version", "locatorStem", "locatorVersion", "fragmentPath")
39
+
40
+
41
+ def _version_tail(spec: Spec, type_: str) -> bool:
42
+ """Whether `dispatch.<type>.versionTail` is set. Read from the spec, never guessed from punctuation."""
43
+ dispatch = spec.get_object("dispatch") or {}
44
+ entry = dispatch.get(type_)
45
+ return isinstance(entry, dict) and entry.get("versionTail") is True
46
+
47
+
48
+ def _split(spec: Spec, type_: str, locator: str) -> tuple[str, str | None]:
49
+ """The locator without its version, and the version it carried — only for a type whose dispatch entry
50
+ declares `versionTail`.
51
+
52
+ **Whether a locator carries a version at all is the type's business**, read from `versionTail` rather
53
+ than guessed from punctuation. Within a type that does carry one, an `@` that opens a segment —
54
+ preceded by `/`, or first — belongs to a namespace (`npm/@acme/x` has no version). An `@` inside a
55
+ segment closes the name: the version runs from it to the next `/`, and whatever follows that `/` is a
56
+ path inside the named thing. **The path stays in the stem, and only the version leaves it** — a file
57
+ at two releases is one file, so `npm/x@1.0.0/docs/guide.md` and `npm/x@2.0.0/docs/guide.md` share the
58
+ stem `npm/x/docs/guide.md`. Ported from `crates/ref-id/src/relations.rs`'s `split`.
59
+ """
60
+ if not _version_tail(spec, type_):
61
+ return locator, None
62
+ for index in range(1, len(locator)):
63
+ if locator[index] != "@" or locator[index - 1] == "/":
64
+ continue
65
+ slash = locator.find("/", index)
66
+ if slash < 0:
67
+ return locator[:index], locator[index + 1 :]
68
+ return locator[:index] + locator[slash:], locator[index + 1 : slash]
69
+ return locator, None
70
+
71
+
72
+ def _stem_reaches(general: str, specific: str) -> bool:
73
+ """Whether the general identifier's stem reaches the specific one's — equal, or a whole **segment**
74
+ prefix of it. Ported from `crates/ref-id/src/relations.rs`'s `stem_reaches`."""
75
+ return general == specific or specific.startswith(f"{general}/")
76
+
77
+
78
+ def _read(spec: Spec, identifier: IdentifierOrParsed) -> ParseResult | None:
79
+ """The identifier, when this package vouches for how it was decomposed — `ok` or `uncovered` only.
80
+ `malformed` has no decomposition to compare, and `unsupported` has one read by the wrong grammar."""
81
+ parsed = identifier if isinstance(identifier, ParseResult) else parse(identifier)
82
+ if parsed.status in (spec.status("ok"), spec.status("uncovered")):
83
+ return parsed
84
+ return None
85
+
86
+
87
+ def _nested_value(parsed: ParseResult, key: str) -> str | None:
88
+ """The decoded nested identifier behind a qualifier's raw value, keyed by qualifier key — `None` when
89
+ this key's value is not one, on this side."""
90
+ if parsed.nested is None:
91
+ return None
92
+ for candidate_key, value in parsed.nested:
93
+ if candidate_key == key:
94
+ return value
95
+ return None
96
+
97
+
98
+ def _optional_relation(a: str | None, b: str | None) -> Relation:
99
+ """The four-case relation between two optional raw values: equal when both agree or neither declares;
100
+ covers when only the second declares; coveredBy for the mirror; differ when both declare and
101
+ disagree."""
102
+ if a is None and b is None:
103
+ return "equal"
104
+ if a is None:
105
+ return "covers"
106
+ if b is None:
107
+ return "coveredBy"
108
+ return "equal" if a == b else "differ"
109
+
110
+
111
+ def _segment_relation(a: str, b: str) -> Relation:
112
+ """The locator-stem relation: equal, a whole-segment prefix either way (`covers`/`coveredBy`), or
113
+ differ."""
114
+ if a == b:
115
+ return "equal"
116
+ if _stem_reaches(a, b):
117
+ return "covers"
118
+ if _stem_reaches(b, a):
119
+ return "coveredBy"
120
+ return "differ"
121
+
122
+
123
+ def _qualifier_relation(
124
+ spec: Spec, x: ParseResult, y: ParseResult, key: str, xv: str | None, yv: str | None
125
+ ) -> QualifierRelation:
126
+ """One qualifier's relation: raw-value comparison, except where both sides' value is a nested `ref:`
127
+ identifier this relation accepts, where the nested pair is related the same way, one level deep, and
128
+ this member's own relation is that nested result reduced (`_reduce`)."""
129
+ if xv is not None and yv is not None:
130
+ xa = _nested_value(x, key)
131
+ ya = _nested_value(y, key)
132
+ if xa is not None and ya is not None:
133
+ xp = _read(spec, xa)
134
+ yp = _read(spec, ya)
135
+ if xp is not None and yp is not None:
136
+ nested = _relate_result(spec, xp, yp)
137
+ return QualifierRelation(relation=_reduce(nested), nested=nested)
138
+ return QualifierRelation(relation=_optional_relation(xv, yv))
139
+
140
+
141
+ def _declared_keys(x: tuple[Pair, ...], y: tuple[Pair, ...]) -> list[str]:
142
+ """The keys of two pair tuples, each once, in first-seen order across both sides.
143
+
144
+ `pair_key not in keys` against a growing list is quadratic in the qualifier count; a set kept beside
145
+ the ordered list turns that membership check constant while `keys` still carries the order."""
146
+ keys: list[str] = []
147
+ seen: set[str] = set()
148
+ for pair_key, _pair_value in x + y:
149
+ if pair_key not in seen:
150
+ seen.add(pair_key)
151
+ keys.append(pair_key)
152
+ return keys
153
+
154
+
155
+ def _relate_result(spec: Spec, x: ParseResult, y: ParseResult) -> RelateResult:
156
+ """`comparison.relate.result`: every dimension computed on its own, whatever the others found."""
157
+ type_relation: Relation = "equal" if x.type == y.type else "differ"
158
+ version_relation: Relation = "equal" if x.version == y.version else "differ"
159
+
160
+ x_stem, x_version = _split(spec, x.type, x.locator)
161
+ y_stem, y_version = _split(spec, y.type, y.locator)
162
+ locator_stem = _segment_relation(x_stem, y_stem)
163
+ locator_version = _optional_relation(x_version, y_version)
164
+
165
+ x_path = x.fragment.path if x.fragment is not None else None
166
+ y_path = y.fragment.path if y.fragment is not None else None
167
+ fragment_path = _optional_relation(x_path, y_path)
168
+
169
+ xr = x.fragment.refinements if x.fragment is not None else ()
170
+ yr = y.fragment.refinements if y.fragment is not None else ()
171
+ xr_map = dict(xr)
172
+ yr_map = dict(yr)
173
+ fragment_refinements = tuple(
174
+ (key, _optional_relation(xr_map.get(key), yr_map.get(key))) for key in _declared_keys(xr, yr)
175
+ )
176
+
177
+ x_map = dict(x.qualifiers)
178
+ y_map = dict(y.qualifiers)
179
+ qualifiers = tuple(
180
+ (key, _qualifier_relation(spec, x, y, key, x_map.get(key), y_map.get(key)))
181
+ for key in _declared_keys(x.qualifiers, y.qualifiers)
182
+ )
183
+
184
+ return RelateResult(
185
+ type=type_relation,
186
+ version=version_relation,
187
+ locator_stem=locator_stem,
188
+ locator_version=locator_version,
189
+ fragment_path=fragment_path,
190
+ fragment_refinements=fragment_refinements,
191
+ qualifiers=qualifiers,
192
+ )
193
+
194
+
195
+ def _reduce(result: RelateResult) -> Relation:
196
+ """`comparison.relate.reduction`: a result reduces to one relation across the five fixed dimensions,
197
+ each refinement, and each qualifier's own relation (which, for a nested qualifier, is already that
198
+ nested result reduced by this same rule)."""
199
+ relations: list[Relation] = [
200
+ result.type,
201
+ result.version,
202
+ result.locator_stem,
203
+ result.locator_version,
204
+ result.fragment_path,
205
+ ]
206
+ relations.extend(relation for _key, relation in result.fragment_refinements)
207
+ relations.extend(qualifier.relation for _key, qualifier in result.qualifiers)
208
+ if all(relation == "equal" for relation in relations):
209
+ return "equal"
210
+ if all(relation in ("equal", "covers") for relation in relations):
211
+ return "covers"
212
+ if all(relation in ("equal", "coveredBy") for relation in relations):
213
+ return "coveredBy"
214
+ return "differ"
215
+
216
+
217
+ def _same_package_reduces(result: RelateResult) -> bool:
218
+ """`comparison.relate.reductions.samePackage`: type, version, locator stem, fragment path and every
219
+ refinement equal; the locator version ignored entirely; each qualifier equal, or — where it carries a
220
+ nested result — `samePackage` holding on that nested result instead of its own (possibly `differ`)
221
+ relation, so one engine at two releases is one engine even though `by=` itself reports `differ`."""
222
+ if result.type != "equal" or result.version != "equal" or result.locator_stem != "equal" or result.fragment_path != "equal":
223
+ return False
224
+ if any(relation != "equal" for _key, relation in result.fragment_refinements):
225
+ return False
226
+ for _key, qualifier in result.qualifiers:
227
+ if qualifier.nested is not None:
228
+ if not _same_package_reduces(qualifier.nested):
229
+ return False
230
+ elif qualifier.relation != "equal":
231
+ return False
232
+ return True
233
+
234
+
235
+ def same_identifier(a: IdentifierOrParsed, b: IdentifierOrParsed, /) -> bool:
236
+ """Whether two identifiers name one thing: their canonical spellings are equal byte for byte
237
+ (`identifierEquivalence.comparison`). An identifier with no canonical form — malformed, or at a scheme
238
+ version this package does not support — names nothing here, so it is never the same as anything,
239
+ including itself; the same status check `_read` uses (`ok` or `uncovered` only) gates this before
240
+ `canonical_identifier` ever runs."""
241
+ spec = load_spec()
242
+ x = _read(spec, a)
243
+ y = _read(spec, b)
244
+ if x is None or y is None:
245
+ return False
246
+ try:
247
+ return canonical_identifier(x) == canonical_identifier(y)
248
+ except RefIdError:
249
+ return False
250
+
251
+
252
+ def relate(a: IdentifierOrParsed, b: IdentifierOrParsed, /) -> RelateResult | None:
253
+ """How two identifiers relate in each dimension the specification names — `comparison.relate`. `None`
254
+ for a pair this relation refuses: malformed, or at a scheme version this package does not support, on
255
+ either side. `covers`, `same_package` and `covers(b, a)` (`coveredBy`) are reductions of this result."""
256
+ spec = load_spec()
257
+ x = _read(spec, a)
258
+ y = _read(spec, b)
259
+ if x is None or y is None:
260
+ return None
261
+ return _relate_result(spec, x, y)
262
+
263
+
264
+ def same_package(a: IdentifierOrParsed, b: IdentifierOrParsed, /) -> bool:
265
+ """Whether two identifiers name the same released thing, at whatever version each declares.
266
+
267
+ Symmetric, and version-blind in exactly one place: the locator of a type whose dispatch entry declares
268
+ `versionTail`. Everything else still distinguishes. A malformed identifier, and one at an identifier
269
+ version this package does not implement, name nothing here and so are the same as nothing, including
270
+ themselves. Implemented as `comparison.relate.reductions.samePackage` on `relate`'s result, so it
271
+ cannot disagree with `relate` about the same pair."""
272
+ spec = load_spec()
273
+ x = _read(spec, a)
274
+ y = _read(spec, b)
275
+ if x is None or y is None:
276
+ return False
277
+ return _same_package_reduces(_relate_result(spec, x, y))
278
+
279
+
280
+ def covers(general: IdentifierOrParsed, specific: IdentifierOrParsed, /) -> bool:
281
+ """Whether the first identifier is the second with less declared — the general covering the specific.
282
+
283
+ Asymmetric, and the direction is the whole point. Implemented as `comparison.relate.reductions.covers`
284
+ on `relate`'s result, so it cannot disagree with `relate` about the same pair."""
285
+ spec = load_spec()
286
+ x = _read(spec, general)
287
+ y = _read(spec, specific)
288
+ if x is None or y is None:
289
+ return False
290
+ return _reduce(_relate_result(spec, x, y)) in ("equal", "covers")
291
+
292
+
293
+ def _qualifier_verdict_axis(spec: Spec, key: str) -> str | None:
294
+ """A qualifier's `verdict.axis`, read from `spec.qualifiers.<key>.verdict` — never restated
295
+ (`.agents/rules/repo-guardrails.md`). `None` for a key with no `verdict` member."""
296
+ value = spec.value("qualifiers", key, "verdict", "axis")
297
+ return value if isinstance(value, str) else None
298
+
299
+
300
+ def _qualifier_verdict_conflict(spec: Spec, key: str) -> str | None:
301
+ """A qualifier's `verdict.conflict`, or `None` when the key has no `verdict` member at all."""
302
+ value = spec.value("qualifiers", key, "verdict", "conflict")
303
+ return value if isinstance(value, str) else None
304
+
305
+
306
+ def _qualifier_verdict_fallback(spec: Spec, key: str) -> tuple[list[str], str] | None:
307
+ """A qualifier's `verdict.conflictWhenNeitherSideDeclares` (`keys`, `then`), when declared."""
308
+ node = spec.value("qualifiers", key, "verdict", "conflictWhenNeitherSideDeclares")
309
+ if not isinstance(node, dict):
310
+ return None
311
+ keys = node.get("keys")
312
+ then = node.get("then")
313
+ if not isinstance(keys, list) or not isinstance(then, str):
314
+ return None
315
+ return [key for key in keys if isinstance(key, str)], then
316
+
317
+
318
+ def _fixed_relation(result: RelateResult, dimension: str) -> Relation:
319
+ """One of the five fixed dimensions of a `RelateResult`, read by its `decidedBy` path name rather than
320
+ by attribute — the path names are the specification's vocabulary, the dataclass fields are Python's."""
321
+ if dimension == "type":
322
+ return result.type
323
+ if dimension == "version":
324
+ return result.version
325
+ if dimension == "locatorStem":
326
+ return result.locator_stem
327
+ if dimension == "locatorVersion":
328
+ return result.locator_version
329
+ if dimension == "fragmentPath":
330
+ return result.fragment_path
331
+ raise AssertionError(f"_fixed_relation called with a non-fixed dimension: {dimension}")
332
+
333
+
334
+ def _by_key(path: str) -> bytes:
335
+ return path.encode("utf-16-be", "surrogatepass")
336
+
337
+
338
+ def _order_decided(paths: list[str]) -> tuple[str, ...]:
339
+ """`decidedBy`'s fixed order: the five dimensions in `_FIXED_DIMENSIONS`'s order, then refinement
340
+ keys, then qualifier keys, each of the latter two groups sorted by UTF-16 code unit order."""
341
+ fixed = [dimension for dimension in _FIXED_DIMENSIONS if dimension in paths]
342
+ refinements = sorted((path for path in paths if path.startswith("fragmentRefinements.")), key=_by_key)
343
+ qualifiers = sorted((path for path in paths if path.startswith("qualifiers.")), key=_by_key)
344
+ return tuple(fixed + refinements + qualifiers)
345
+
346
+
347
+ def verdict(a: IdentifierOrParsed, b: IdentifierOrParsed, /) -> VerdictResult | None:
348
+ """What two identifiers mean together once location qualifiers are hints rather than identity —
349
+ `relate`'s result reduced onto an identity axis and a content axis, per `comparison.verdict.rule`.
350
+
351
+ `None` for a pair `relate` refuses. Mirrored: `verdict(b, a)` is this result with `covers` and
352
+ `coveredBy` exchanged on the identity axis; the content axis and `decidedBy` are unchanged
353
+ (`comparison.verdict.symmetry`). Ported from `crates/ref-id/src/relations.rs`'s `verdict`.
354
+ """
355
+ spec = load_spec()
356
+ related = relate(a, b)
357
+ if related is None:
358
+ return None
359
+ x = _read(spec, a)
360
+ y = _read(spec, b)
361
+ if x is None or y is None:
362
+ return None
363
+ x_keys = {key for key, _value in x.qualifiers}
364
+ y_keys = {key for key, _value in y.qualifiers}
365
+
366
+ # The content axis: each qualifier whose verdict.axis is "content" (spec.qualifiers.*.verdict).
367
+ content_equal: list[str] = []
368
+ content_differ: list[str] = []
369
+ content_one_side: list[str] = []
370
+ for key, qualifier in related.qualifiers:
371
+ if _qualifier_verdict_axis(spec, key) != "content":
372
+ continue
373
+ path = f"qualifiers.{key}"
374
+ if qualifier.relation == "equal":
375
+ content_equal.append(path)
376
+ elif qualifier.relation == "differ":
377
+ content_differ.append(path)
378
+ else: # covers or coveredBy: declared on one side only
379
+ content_one_side.append(path)
380
+
381
+ content: VerdictContent
382
+ content_decided: list[str]
383
+ if content_differ:
384
+ content, content_decided = "different", content_differ
385
+ elif content_equal:
386
+ content, content_decided = "same", content_equal
387
+ else:
388
+ content, content_decided = "unknown", content_one_side
389
+
390
+ # The identity axis. Step one: the five fixed dimensions that relate as "differ".
391
+ distinct: list[str] = [dimension for dimension in _FIXED_DIMENSIONS if _fixed_relation(related, dimension) == "differ"]
392
+
393
+ # Step two: each identity-axis qualifier with a declared conflict, whose relation is "differ", decides
394
+ # by its conflict — or by conflictWhenNeitherSideDeclares.then when none of its listed keys is
395
+ # declared on either side ("declared" meaning present among that side's own parsed qualifiers).
396
+ # Step three: a qualifier with no verdict member, or a fragment refinement (which never has one),
397
+ # relating as "differ" makes identity distinct outright.
398
+ undetermined: list[str] = []
399
+ for key, qualifier in related.qualifiers:
400
+ if _qualifier_verdict_axis(spec, key) == "content":
401
+ continue
402
+ path = f"qualifiers.{key}"
403
+ if qualifier.relation != "differ":
404
+ continue
405
+ conflict = _qualifier_verdict_conflict(spec, key)
406
+ if conflict is None:
407
+ distinct.append(path) # step three
408
+ continue
409
+ decision = conflict # step two, default
410
+ fallback = _qualifier_verdict_fallback(spec, key)
411
+ if fallback is not None:
412
+ fallback_keys, then = fallback
413
+ if not any(fallback_key in x_keys or fallback_key in y_keys for fallback_key in fallback_keys):
414
+ decision = then
415
+ if decision == "distinct":
416
+ distinct.append(path)
417
+ else:
418
+ undetermined.append(path)
419
+ for key, relation in related.fragment_refinements:
420
+ if relation == "differ":
421
+ distinct.append(f"fragmentRefinements.{key}")
422
+
423
+ identity: VerdictIdentity
424
+ identity_decided: list[str]
425
+ if distinct:
426
+ identity, identity_decided = "distinct", distinct
427
+ elif undetermined:
428
+ # Step four: an undetermined from step two, failing a distinct, makes identity undetermined —
429
+ # decided by the conflicting location keys alone.
430
+ identity, identity_decided = "undetermined", undetermined
431
+ else:
432
+ # Step five: relate's result reduced with the content-axis qualifiers set aside. Every member
433
+ # reaching here relates as "equal", "covers" or "coveredBy" — a "differ" would already have been
434
+ # caught by steps one through three.
435
+ members: list[tuple[str, Relation]] = [(dimension, _fixed_relation(related, dimension)) for dimension in _FIXED_DIMENSIONS]
436
+ members.extend((f"fragmentRefinements.{key}", relation) for key, relation in related.fragment_refinements)
437
+ for key, qualifier in related.qualifiers:
438
+ if _qualifier_verdict_axis(spec, key) == "content":
439
+ continue
440
+ members.append((f"qualifiers.{key}", qualifier.relation))
441
+ has_covers = any(relation == "covers" for _path, relation in members)
442
+ has_covered_by = any(relation == "coveredBy" for _path, relation in members)
443
+ if not has_covers and not has_covered_by:
444
+ identity, identity_decided = "same", []
445
+ elif has_covers and not has_covered_by:
446
+ identity = "covers"
447
+ identity_decided = [path for path, relation in members if relation == "covers"]
448
+ elif has_covered_by and not has_covers:
449
+ identity = "coveredBy"
450
+ identity_decided = [path for path, relation in members if relation == "coveredBy"]
451
+ else:
452
+ # Neither reaches the other: one side declares what the other leaves open in one place and
453
+ # the reverse in another, so nothing separates them.
454
+ identity = "undetermined"
455
+ identity_decided = [path for path, relation in members if relation != "equal"]
456
+
457
+ return VerdictResult(
458
+ identity=identity,
459
+ content=content,
460
+ decided_by=VerdictDecidedBy(
461
+ identity=_order_decided(identity_decided),
462
+ content=_order_decided(content_decided),
463
+ ),
464
+ )
ref_id/serialise.py ADDED
@@ -0,0 +1,109 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """Reassembles a `ParseResult` into the exact bytes it was parsed from. Every part is emitted verbatim.
3
+
4
+ Ported from `crates/ref-id/src/serialise.rs`; `canonical_identifier` from `crates/ref-id/src/relations.rs`'s
5
+ `canonical_form`.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ from dataclasses import replace
10
+
11
+ from .encoding import encode, table_for
12
+ from .errors import SerialiseError
13
+ from .grammar import FIELD, FRAGMENT_INTRODUCER, PAIR, scheme_prefix
14
+ from .parse import parse
15
+ from .spec import Spec, load_spec
16
+ from .types import Fragment, ParseResult
17
+
18
+ __all__ = ["canonical_identifier", "serialise"]
19
+
20
+
21
+ def _pairs(entries: tuple[tuple[str, str], ...], separator: str) -> str:
22
+ return separator.join(f"{key}{PAIR}{value}" for key, value in entries)
23
+
24
+
25
+ def serialise(parsed: ParseResult, /) -> str:
26
+ """`serialise(parse(s)) == s` for every parseable input, including uncovered and unsupported ones. A
27
+ malformed result has no faithful form and is refused."""
28
+ spec = load_spec()
29
+ if parsed.status == spec.status("malformed"):
30
+ raise SerialiseError(
31
+ spec.part(parsed.part) if parsed.part is not None else spec.part("grammar"),
32
+ "a malformed identifier cannot be serialised without losing the part that failed",
33
+ )
34
+ out = [scheme_prefix(spec)]
35
+ if parsed.explicit_version:
36
+ out.append(parsed.version_text if parsed.version_text is not None else str(parsed.version))
37
+ out.append(FIELD)
38
+ out.append(parsed.type)
39
+ out.append(FIELD)
40
+ out.append(parsed.locator)
41
+ if parsed.qualifiers:
42
+ separator = spec.get_str("grammar", "state", "separator")
43
+ out.append(separator)
44
+ out.append(_pairs(parsed.qualifiers, separator))
45
+ if parsed.fragment is not None:
46
+ out.append(FRAGMENT_INTRODUCER)
47
+ out.append(parsed.fragment.path)
48
+ if parsed.fragment.refinements:
49
+ separator = spec.get_str("grammar", "fragment", "separator")
50
+ out.append(separator)
51
+ out.append(_pairs(parsed.fragment.refinements, separator))
52
+ return "".join(out)
53
+
54
+
55
+ def _by_key(pair: tuple[str, str]) -> bytes:
56
+ return pair[0].encode("utf-16-be", "surrogatepass")
57
+
58
+
59
+ def _nesting_form(spec: Spec, key: str) -> dict[str, object] | None:
60
+ """The qualifier's declared form that nests an identifier, if it declares one — ported from
61
+ `crates/ref-id/src/relations.rs`'s `nesting_form`."""
62
+ declared = spec.get_object("qualifiers")
63
+ if declared is None:
64
+ return None
65
+ entry = declared.get(key)
66
+ if not isinstance(entry, dict):
67
+ return None
68
+ forms = spec.get_object("forms") or {}
69
+ for name in entry.get("forms", []):
70
+ if not isinstance(name, str):
71
+ continue
72
+ form = forms.get(name)
73
+ if isinstance(form, dict) and form.get("nested") is True:
74
+ return form
75
+ return None
76
+
77
+
78
+ def _canonical_form(spec: Spec, parsed: ParseResult) -> str:
79
+ """`identifierEquivalence.canonicalForm`: qualifiers and the fragment's refinements sorted by key
80
+ (UTF-16 code unit order); a nested `ref:` identifier inside a qualifier value re-written in its own
81
+ canonical form (decode, canonicalise, re-encode); the version slot omitted when it holds
82
+ `version.default`. Every other part is left exactly as parsed.
83
+
84
+ A malformed identifier has no canonical form: `serialise` refuses it, naming the part that failed.
85
+ """
86
+ out = replace(parsed, qualifiers=tuple(sorted(parsed.qualifiers, key=_by_key)))
87
+ if out.fragment is not None:
88
+ out = replace(out, fragment=Fragment(path=out.fragment.path, refinements=tuple(sorted(out.fragment.refinements, key=_by_key))))
89
+ if parsed.nested is not None:
90
+ qualifiers = list(out.qualifiers)
91
+ for key, raw in parsed.nested:
92
+ form = _nesting_form(spec, key)
93
+ if form is None:
94
+ continue
95
+ inner = parse(raw)
96
+ canonical_inner = _canonical_form(spec, inner)
97
+ encoded = encode(canonical_inner, table_for(spec, form))
98
+ qualifiers = [(k, encoded) if k == key else (k, v) for k, v in qualifiers]
99
+ out = replace(out, qualifiers=tuple(qualifiers))
100
+ if out.version == spec.get_int("version", "default"):
101
+ out = replace(out, explicit_version=False)
102
+ return serialise(out)
103
+
104
+
105
+ def canonical_identifier(identifier: str | ParseResult, /) -> str:
106
+ """The canonical spelling of an identifier — `identifierEquivalence.canonicalForm`."""
107
+ parsed = identifier if isinstance(identifier, ParseResult) else parse(identifier)
108
+ spec = load_spec()
109
+ return _canonical_form(spec, parsed)