@ccoalm/ccl-skills 0.7.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (127) hide show
  1. package/README.md +2 -2
  2. package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/SKILL.md +8 -7
  3. package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/references/mobile-quality-release.md +6 -1
  4. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +16 -17
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/client-routing.md +1 -1
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/manual-invocation-and-prompts.md +6 -0
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +195 -7
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/timeout-auth-and-capabilities.md +3 -3
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/claude_review.sh +13 -5
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh +9 -3
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/kimi_review.sh +9 -3
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/normalize_review_timeout.sh +22 -0
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/opencode_review.sh +9 -3
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +1540 -129
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_claude_review_probe.sh +8 -3
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +76 -1
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +1858 -3
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_update_review_plan_intent.sh +789 -0
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/update_review_plan_intent.py +513 -0
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +1 -0
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/feature-risk-router/SKILL.md +3 -1
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/architecture-playbook.md +1 -1
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/multi-tenant-isolation.md +1 -1
  24. package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/SKILL.md +4 -1
  25. package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/state-machine-task-patterns.md +2 -0
  26. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/SKILL.md +2 -1
  27. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/inference-capacity-operations.md +24 -0
  28. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/llm-client-gateway.md +1 -1
  29. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/model-prompt-evaluation.md +4 -1
  30. package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +11 -10
  31. package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/references/contracts-and-state.md +5 -0
  32. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/SKILL.md +64 -0
  33. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/agents/openai.yaml +4 -0
  34. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/async-lifecycle-and-performance.md +73 -0
  35. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/runtime-and-project-contract.md +58 -0
  36. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/source-map.md +41 -0
  37. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/verification-diagnostics-and-security.md +63 -0
  38. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +3 -2
  39. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/metrics-conventions.md +8 -1
  40. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/sli-slo-design.md +2 -2
  41. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/canary-and-rollout-strategy.md +16 -2
  42. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/promotion-gate-and-review.md +9 -0
  43. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +14 -16
  44. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/code-review-checklist.md +4 -0
  45. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/delivery-lifecycle.md +1 -1
  46. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/design-routing-and-readiness.md +10 -14
  47. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md +1 -0
  48. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/verify-developer-experience.md +1 -1
  49. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/SKILL.md +135 -86
  50. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/behavioral-aesthetic-logic.md +66 -80
  51. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/delivery-contract.md +275 -0
  52. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-execution-checklist.md +88 -214
  53. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-impl-naming-and-versioning.md +2 -2
  54. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-intake-and-acceptance.md +10 -8
  55. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-system-source-of-truth.md +6 -5
  56. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md +112 -95
  57. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/frontend-code-evidence-map.md +30 -21
  58. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/interaction-design-patterns.md +22 -3
  59. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/layout-recipes-and-screenshot-acceptance.md +20 -17
  60. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/multi-project-token-consistency.md +7 -9
  61. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/multi-stack-strategy.md +14 -10
  62. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/operational-processing-workflows.md +2 -0
  63. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/platform-mobile-patterns.md +3 -3
  64. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/product-lifecycle-acceptance-and-iteration.md +9 -6
  65. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/product-surface-patterns.md +3 -0
  66. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/source-map.md +37 -10
  67. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/tokens-and-components.md +8 -1
  68. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/ui-ux-audit.md +16 -5
  69. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/ui-ux-design-development.md +16 -5
  70. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/visual-craft.md +4 -2
  71. package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/multi-tenant-isolation.md +1 -1
  72. package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/SKILL.md +4 -1
  73. package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/state-machine-task-patterns.md +2 -0
  74. package/dist/assets/marketplace/plugins/ccl-skills/skills/release-coordination/SKILL.md +1 -1
  75. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +8 -8
  76. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/description-authoring.md +4 -0
  77. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +104 -5
  78. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +11 -9
  79. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/r0-leakage-audit.md +102 -0
  80. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +103 -0
  81. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-to-skill-extraction.md +20 -0
  82. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/uiux-judgment-extraction.md +6 -6
  83. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +4 -3
  84. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-ccl-skills.sh +69 -2
  85. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +22 -0
  86. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +49 -4
  87. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/obligation-ledger.py +2748 -0
  88. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/register-firing-path-resolution.rb +20 -5
  89. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/shared_git_surface_gate.py +1142 -0
  90. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +17 -0
  91. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_skill_catalog.sh +41 -4
  92. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ci_checkout_ref_binding.sh +120 -0
  93. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_entrypoint_domain_scan_terms.sh +82 -8
  94. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +336 -0
  95. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_self_adjudication.sh +82 -10
  96. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_obligation_ledger.sh +1416 -0
  97. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_obligation_ledger_repo_audit.sh +57 -0
  98. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_wiring.sh +141 -4
  99. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_routing_pointer_integrity.sh +3 -1
  100. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_shared_git_surface_gate.sh +1696 -0
  101. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_uiux_delivery_contract.sh +2117 -0
  102. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_uiux_loading_budget.sh +316 -0
  103. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +1176 -0
  104. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_skill_cross_refs.sh +31 -1
  105. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate-skill.sh +9 -4
  106. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +980 -0
  107. package/dist/assets/marketplace/plugins/ccl-skills/skills/terminal-cli-dev/SKILL.md +8 -6
  108. package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/classical-test-design-techniques.md +1 -1
  109. package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/tc-review-and-prioritization.md +1 -1
  110. package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/update-lifecycle.md +2 -0
  111. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/SKILL.md +16 -15
  112. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/ci-fixtures-and-flake-control.md +5 -1
  113. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/client-runtime-test-matrices.md +10 -2
  114. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/e2e-real-flow-testing.md +2 -2
  115. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/integration-contract-testing.md +10 -0
  116. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/test-code-authoring-patterns.md +2 -2
  117. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/test-topology-and-commands.md +1 -1
  118. package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/SKILL.md +2 -1
  119. package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/annotation-driven-revision.md +9 -0
  120. package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/figure-and-table-craft.md +8 -2
  121. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/SKILL.md +7 -5
  122. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/complex-workspace-patterns.md +1 -1
  123. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/react-architecture.md +3 -0
  124. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/web-quality-release.md +37 -4
  125. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/web-ui-quality.md +10 -1
  126. package/dist/assets/release.json +215 -105
  127. package/package.json +1 -1
@@ -0,0 +1,2748 @@
1
+ #!/usr/bin/env python3
2
+ """Generate and audit a closed obligation-preservation ledger.
3
+
4
+ The row set is derived from every pre-existing ``skills/**/*.md`` path changed
5
+ against an explicit base revision. Humans bind each derived row to one exact
6
+ current carrier in a JSONL mapping; this program never guesses or fuzzily binds
7
+ carriers. Line ranges in the reader-facing ledger are derived from the current
8
+ files, so stale locators are detectable without putting a dirty-tree digest in
9
+ the generated document.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import argparse
15
+ import hashlib
16
+ import importlib.util
17
+ import json
18
+ import re
19
+ import subprocess
20
+ import sys
21
+ from collections import Counter
22
+ from dataclasses import dataclass
23
+ from pathlib import Path
24
+ from typing import Any, Iterable
25
+
26
+
27
+ SCHEMA_VERSION = 3
28
+ SCHEMA_VERSIONS = {3, 4}
29
+ DISPOSITIONS = {
30
+ "merged",
31
+ "subsumed",
32
+ "rehosted",
33
+ "retired-dead",
34
+ "partitioned",
35
+ "partial-retirement",
36
+ }
37
+ EFFECTS = {"preserved", "strengthened", "retired", "unresolved"}
38
+ QUALIFIER_KINDS = {
39
+ "modality",
40
+ "recency",
41
+ "threshold",
42
+ "scope",
43
+ "actor",
44
+ "consequence",
45
+ }
46
+ MAPPING_FIELDS = {
47
+ "schema_version",
48
+ "source_path",
49
+ "source_ordinal",
50
+ "reason",
51
+ "before_text",
52
+ "before_chain",
53
+ "disposition",
54
+ "effect",
55
+ "carrier_path",
56
+ "carrier_text",
57
+ "carrier_chain",
58
+ "carrier_bundle",
59
+ "compound_clauses",
60
+ "manual_reviewed",
61
+ "semantic_review",
62
+ "semantic_rationale",
63
+ "qualifiers",
64
+ "qualifier_relations",
65
+ "review_note",
66
+ "_mapping_line",
67
+ }
68
+ MAPPING_FIELDS_V4 = {
69
+ "schema_version",
70
+ "source_path",
71
+ "source_ordinal",
72
+ "reason",
73
+ "before_text",
74
+ "before_chain",
75
+ "disposition",
76
+ "effect",
77
+ "parts",
78
+ "manual_reviewed",
79
+ "semantic_review",
80
+ "semantic_rationale",
81
+ "review_note",
82
+ "_mapping_line",
83
+ }
84
+ REASONS = {"left-a-rewritten-line", "governing-chain-changed"}
85
+ PROVENANCE_BASENAMES = {"source-register.md", "provenance.md"}
86
+
87
+
88
+ def _load_chain_diff(script_dir: Path) -> Any:
89
+ path = script_dir / "governing-chain-diff.py"
90
+ spec = importlib.util.spec_from_file_location("governing_chain_diff", path)
91
+ if spec is None or spec.loader is None:
92
+ raise RuntimeError(f"cannot import {path}")
93
+ module = importlib.util.module_from_spec(spec)
94
+ sys.modules[spec.name] = module
95
+ spec.loader.exec_module(module)
96
+ return module
97
+
98
+
99
+ GCD = _load_chain_diff(Path(__file__).resolve().parent)
100
+
101
+
102
+ class AuditError(Exception):
103
+ def __init__(self, code: str, detail: str):
104
+ super().__init__(detail)
105
+ self.code = code
106
+ self.detail = detail
107
+
108
+
109
+ @dataclass(frozen=True)
110
+ class ExpectedRow:
111
+ source_path: str
112
+ source_ordinal: int
113
+ reason: str
114
+ before_text: str
115
+ before_chain: tuple[str, ...]
116
+
117
+ @property
118
+ def key(self) -> tuple[str, int]:
119
+ return (self.source_path, self.source_ordinal)
120
+
121
+
122
+ @dataclass(frozen=True)
123
+ class Carrier:
124
+ path: str
125
+ text: str
126
+ chain: tuple[str, ...]
127
+ start_line: int
128
+ end_line: int
129
+
130
+
131
+ @dataclass(frozen=True)
132
+ class PartitionedPart:
133
+ part_id: str
134
+ status: str
135
+ source_text: str
136
+ carriers: tuple[Carrier, ...]
137
+
138
+
139
+ @dataclass(frozen=True)
140
+ class LedgerObligation:
141
+ text: str
142
+ chain: tuple[str, ...]
143
+
144
+ @property
145
+ def key(self) -> str:
146
+ return GCD.normalize(self.text)
147
+
148
+
149
+ def git(repo: Path, *args: str, check: bool = True) -> subprocess.CompletedProcess[str]:
150
+ result = subprocess.run(
151
+ ["git", "-C", str(repo), *args], capture_output=True, text=True
152
+ )
153
+ if check and result.returncode != 0:
154
+ raise AuditError("GIT_FAILED", result.stderr.strip() or "git command failed")
155
+ return result
156
+
157
+
158
+ def normalize_path(value: str) -> str:
159
+ path = Path(value)
160
+ if path.is_absolute() or ".." in path.parts:
161
+ raise AuditError("INVALID_PATH", f"path must be repository-relative: {value}")
162
+ return path.as_posix()
163
+
164
+
165
+ def changed_preexisting_paths(
166
+ repo: Path, base: str, head: str | None = None
167
+ ) -> list[str]:
168
+ diff_args = ["diff", "--name-only", "--diff-filter=ACDMRTUXB", base]
169
+ if head is not None:
170
+ diff_args.append(head)
171
+ changed = git(
172
+ repo,
173
+ *diff_args,
174
+ "--",
175
+ "skills",
176
+ ).stdout.splitlines()
177
+ out: list[str] = []
178
+ for raw in changed:
179
+ path = normalize_path(raw)
180
+ if not path.startswith("skills/") or not path.endswith(".md"):
181
+ continue
182
+ exists_at_base = git(repo, "cat-file", "-e", f"{base}:{path}", check=False)
183
+ if exists_at_base.returncode == 0:
184
+ out.append(path)
185
+ return sorted(set(out))
186
+
187
+
188
+ def read_base(repo: Path, base: str, path: str) -> str:
189
+ return git(repo, "show", f"{base}:{path}").stdout
190
+
191
+
192
+ def read_current(repo: Path, path: str) -> str:
193
+ file_path = repo / path
194
+ return file_path.read_text(encoding="utf-8") if file_path.is_file() else ""
195
+
196
+
197
+ def read_at(repo: Path, rev: str, path: str) -> str:
198
+ result = git(repo, "show", f"{rev}:{path}", check=False)
199
+ return result.stdout if result.returncode == 0 else ""
200
+
201
+
202
+ def derive_rows(
203
+ repo: Path, base: str, paths: Iterable[str], head: str | None = None
204
+ ) -> list[ExpectedRow]:
205
+ rows: list[ExpectedRow] = []
206
+ for path in sorted(paths):
207
+ before = read_base(repo, base, path)
208
+ after = read_current(repo, path) if head is None else read_at(repo, head, path)
209
+ after_by_key: dict[str, list[Any]] = {}
210
+ for obligation, _, _ in parse_obligation_ranges(after):
211
+ after_by_key.setdefault(obligation.key, []).append(obligation)
212
+ ordinal = 0
213
+ for obligation, _, _ in parse_obligation_ranges(before):
214
+ survivors = after_by_key.get(obligation.key)
215
+ reason: str | None = None
216
+ if not survivors:
217
+ reason = "left-a-rewritten-line"
218
+ elif not any(item.chain == obligation.chain for item in survivors):
219
+ reason = "governing-chain-changed"
220
+ if reason is None:
221
+ continue
222
+ ordinal += 1
223
+ rows.append(
224
+ ExpectedRow(
225
+ source_path=path,
226
+ source_ordinal=ordinal,
227
+ reason=reason,
228
+ before_text=GCD.normalize(obligation.text),
229
+ before_chain=tuple(obligation.chain),
230
+ )
231
+ )
232
+ return rows
233
+
234
+
235
+ def load_mapping(path: Path) -> list[dict[str, Any]]:
236
+ rows: list[dict[str, Any]] = []
237
+ if not path.is_file():
238
+ raise AuditError("MAPPING_MISSING", str(path))
239
+ for line_no, raw in enumerate(path.read_text(encoding="utf-8").splitlines(), 1):
240
+ if not raw.strip():
241
+ continue
242
+ try:
243
+ value = json.loads(raw)
244
+ except json.JSONDecodeError as exc:
245
+ raise AuditError("MAPPING_JSON", f"{path}:{line_no}: {exc}") from exc
246
+ if not isinstance(value, dict):
247
+ raise AuditError("MAPPING_JSON", f"{path}:{line_no}: object required")
248
+ value["_mapping_line"] = line_no
249
+ rows.append(value)
250
+ return rows
251
+
252
+
253
+ def mapping_key(row: dict[str, Any]) -> tuple[str, int]:
254
+ try:
255
+ return (normalize_path(str(row["source_path"])), int(row["source_ordinal"]))
256
+ except (KeyError, TypeError, ValueError) as exc:
257
+ raise AuditError("MAPPING_KEY", f"line {row.get('_mapping_line', '?')}: {exc}") from exc
258
+
259
+
260
+ def index_mapping(rows: list[dict[str, Any]]) -> dict[tuple[str, int], dict[str, Any]]:
261
+ indexed: dict[tuple[str, int], dict[str, Any]] = {}
262
+ for row in rows:
263
+ key = mapping_key(row)
264
+ if key in indexed:
265
+ raise AuditError("DUPLICATE_ROW", f"duplicate mapping key {key[0]}#{key[1]}")
266
+ indexed[key] = row
267
+ return indexed
268
+
269
+
270
+ def relocation_destinations(rows: Iterable[dict[str, Any]]) -> list[str]:
271
+ paths: set[str] = set()
272
+ for row in rows:
273
+ carrier_path = row.get("carrier_path")
274
+ if carrier_path:
275
+ path = normalize_path(str(carrier_path))
276
+ if not path.startswith("skills/") or not path.endswith(".md"):
277
+ raise AuditError("CARRIER_OUTSIDE_SKILLS", path)
278
+ paths.add(path)
279
+ bundle = row.get("carrier_bundle")
280
+ if isinstance(bundle, list):
281
+ for member in bundle:
282
+ if not isinstance(member, dict) or not member.get("carrier_path"):
283
+ continue
284
+ path = normalize_path(str(member["carrier_path"]))
285
+ if not path.startswith("skills/") or not path.endswith(".md"):
286
+ raise AuditError("CARRIER_OUTSIDE_SKILLS", path)
287
+ paths.add(path)
288
+ parts = row.get("parts")
289
+ if isinstance(parts, list):
290
+ for part in parts:
291
+ if not isinstance(part, dict):
292
+ continue
293
+ carriers = part.get("carriers")
294
+ if not isinstance(carriers, list):
295
+ continue
296
+ for member in carriers:
297
+ if not isinstance(member, dict) or not member.get("carrier_path"):
298
+ continue
299
+ path = normalize_path(str(member["carrier_path"]))
300
+ if not path.startswith("skills/") or not path.endswith(".md"):
301
+ raise AuditError("CARRIER_OUTSIDE_SKILLS", path)
302
+ paths.add(path)
303
+ return sorted(paths)
304
+
305
+
306
+ def verify_closed_row_set(
307
+ expected: list[ExpectedRow], mapping: dict[tuple[str, int], dict[str, Any]]
308
+ ) -> None:
309
+ expected_by_key = {row.key: row for row in expected}
310
+ missing = sorted(set(expected_by_key) - set(mapping))
311
+ extra = sorted(set(mapping) - set(expected_by_key))
312
+ if missing or extra:
313
+ detail = []
314
+ if missing:
315
+ detail.append("missing=" + ",".join(f"{p}#{n}" for p, n in missing[:12]))
316
+ if extra:
317
+ detail.append("extra=" + ",".join(f"{p}#{n}" for p, n in extra[:12]))
318
+ raise AuditError("ROW_SET_MISMATCH", "; ".join(detail))
319
+ for key, expected_row in expected_by_key.items():
320
+ actual = mapping[key]
321
+ checks = {
322
+ "reason": expected_row.reason,
323
+ "before_text": expected_row.before_text,
324
+ "before_chain": list(expected_row.before_chain),
325
+ }
326
+ for field, wanted in checks.items():
327
+ if actual.get(field) != wanted:
328
+ raise AuditError(
329
+ "ROW_SOURCE_MISMATCH",
330
+ f"{key[0]}#{key[1]} {field}: expected {wanted!r}, got {actual.get(field)!r}",
331
+ )
332
+
333
+
334
+ def normalized_with_offsets(source: str) -> tuple[str, list[int]]:
335
+ chars: list[str] = []
336
+ offsets: list[int] = []
337
+ in_space = False
338
+ for index, char in enumerate(source):
339
+ if char.isspace():
340
+ if chars and not in_space:
341
+ chars.append(" ")
342
+ offsets.append(index)
343
+ in_space = True
344
+ else:
345
+ chars.append(char)
346
+ offsets.append(index)
347
+ in_space = False
348
+ if chars and chars[-1] == " ":
349
+ chars.pop()
350
+ offsets.pop()
351
+ return "".join(chars), offsets
352
+
353
+
354
+ def line_range(source: str, start: int, end: int) -> tuple[int, int]:
355
+ return (source.count("\n", 0, start) + 1, source.count("\n", 0, end) + 1)
356
+
357
+
358
+ def _blank(chars: list[str], start: int, end: int) -> None:
359
+ for index in range(start, end):
360
+ if chars[index] not in "\r\n":
361
+ chars[index] = " "
362
+
363
+
364
+ _FENCE_OPEN = re.compile(r"^[ ]{0,3}(`{3,}|~{3,})([^\r\n]*)$")
365
+ _ATX_HEADING = re.compile(r"^[ ]{0,3}#{1,6}(?:[ \t]|$)")
366
+ _THEMATIC_BREAK = re.compile(r"^[ ]{0,3}([-_*])[ \t]*(?:\1[ \t]*){2,}$")
367
+ _SETEXT_UNDERLINE = re.compile(r"^[ ]{0,3}(?:=+|-+)[ \t]*$")
368
+ _BLOCK_QUOTE_PREFIX = re.compile(r"^[ ]{0,3}(?:>[ ]?)+")
369
+
370
+
371
+ def mask_inert_markdown(source: str) -> str:
372
+ """Blank comments and code containers without changing source offsets.
373
+
374
+ Fence closing follows the relevant CommonMark invariant: same marker,
375
+ closing run at least as long as the opener, and whitespace-only suffix.
376
+ Indented code is also inert — at top level from column 4, inside a list
377
+ item from the content column + 4 — but never where the indented line
378
+ lazily continues an open paragraph; those list continuations remain
379
+ visible to the prose parser.
380
+ """
381
+ chars = list(source)
382
+ comment_start = 0
383
+ while True:
384
+ comment_start = source.find("<!--", comment_start)
385
+ if comment_start < 0:
386
+ break
387
+ comment_end = source.find("-->", comment_start + 4)
388
+ comment_end = len(source) if comment_end < 0 else comment_end + 3
389
+ _blank(chars, comment_start, comment_end)
390
+ comment_start = comment_end
391
+
392
+ commentless = "".join(chars)
393
+ lines = commentless.splitlines(keepends=True)
394
+ offset = 0
395
+ fence_marker: str | None = None
396
+ fence_length = 0
397
+ list_content_column: int | None = None
398
+ paragraph_open = False
399
+ for raw_with_end in lines:
400
+ raw = raw_with_end.rstrip("\r\n")
401
+ line_start = offset
402
+ offset += len(raw_with_end)
403
+
404
+ if fence_marker is not None:
405
+ closer = re.match(
406
+ rf"^[ ]{{0,3}}({re.escape(fence_marker)}{{{fence_length},}})[ \t]*$",
407
+ raw,
408
+ )
409
+ _blank(chars, line_start, offset)
410
+ if closer:
411
+ fence_marker = None
412
+ fence_length = 0
413
+ paragraph_open = False
414
+ continue
415
+
416
+ opener = _FENCE_OPEN.match(raw)
417
+ if opener and not (
418
+ opener.group(1).startswith("`") and "`" in opener.group(2)
419
+ ):
420
+ fence_marker = opener.group(1)[0]
421
+ fence_length = len(opener.group(1))
422
+ _blank(chars, line_start, offset)
423
+ paragraph_open = False
424
+ continue
425
+
426
+ if not raw.strip():
427
+ paragraph_open = False
428
+ continue
429
+ item = GCD._LIST_ITEM.match(raw)
430
+ if item:
431
+ list_content_column = len(raw[: item.start(2)].expandtabs(4))
432
+ # The item's FIRST content line decides the paragraph state: a
433
+ # heading or thematic break there does not open a paragraph, so
434
+ # content-column + 4 indentation after it is code.
435
+ item_text = raw[item.start(2) :]
436
+ paragraph_open = bool(item_text.strip()) and not (
437
+ _ATX_HEADING.match(item_text) or _THEMATIC_BREAK.match(item_text)
438
+ )
439
+ continue
440
+ leading = len(raw) - len(raw.lstrip(" \t"))
441
+ expanded_leading = len(raw[:leading].expandtabs(4))
442
+ if list_content_column is not None and expanded_leading >= list_content_column:
443
+ if expanded_leading - list_content_column >= 4 and not paragraph_open:
444
+ # Indented code inside a list item starts at the content
445
+ # column + 4 and, as at top level, cannot interrupt the
446
+ # item's open paragraph.
447
+ _blank(chars, line_start, offset)
448
+ continue
449
+ content = raw.lstrip(" \t")
450
+ if paragraph_open and _SETEXT_UNDERLINE.match(content):
451
+ paragraph_open = False
452
+ else:
453
+ paragraph_open = not (
454
+ _ATX_HEADING.match(content) or _THEMATIC_BREAK.match(content)
455
+ )
456
+ continue
457
+ if expanded_leading == 0:
458
+ list_content_column = None
459
+ if expanded_leading >= 4 and not paragraph_open:
460
+ # Indented code cannot interrupt a paragraph per CommonMark, but
461
+ # it can start after any other block: blank line, heading, or a
462
+ # closed fence. Continuation lines keep the block open because a
463
+ # masked code line never opens a paragraph.
464
+ _blank(chars, line_start, offset)
465
+ continue
466
+ # Classify the line's block-level content, not its raw prefix: inside
467
+ # a block quote the paragraph state follows the quoted content, so a
468
+ # quoted heading closes paragraph state instead of opening it (while
469
+ # quoted prose still opens it — indented lines after quoted prose are
470
+ # lazy continuations and stay visible).
471
+ content = raw
472
+ quote = _BLOCK_QUOTE_PREFIX.match(content)
473
+ if quote:
474
+ content = content[quote.end() :]
475
+ content_leading = len(content) - len(content.lstrip(" \t"))
476
+ if (
477
+ len(content[:content_leading].expandtabs(4)) >= 4
478
+ and not paragraph_open
479
+ ):
480
+ # Indented code inside a block quote is measured from the
481
+ # quote marker, not the raw line start; the same
482
+ # cannot-interrupt-a-paragraph rule applies.
483
+ _blank(chars, line_start, offset)
484
+ continue
485
+ if paragraph_open and _SETEXT_UNDERLINE.match(content):
486
+ # The line closes an open paragraph as a Setext heading underline,
487
+ # so what follows may open an indented code block.
488
+ paragraph_open = False
489
+ else:
490
+ paragraph_open = bool(content.strip()) and not (
491
+ _ATX_HEADING.match(content) or _THEMATIC_BREAK.match(content)
492
+ )
493
+ return "".join(chars)
494
+
495
+
496
+ def inline_code_only(text: str) -> bool:
497
+ normalized = GCD.normalize(text)
498
+ if not normalized.startswith("`"):
499
+ return False
500
+ opener = len(normalized) - len(normalized.lstrip("`"))
501
+ closer = len(normalized) - len(normalized.rstrip("`"))
502
+ if opener != closer or len(normalized) <= opener + closer:
503
+ return False
504
+ inner = normalized[opener : len(normalized) - closer]
505
+ return bool(inner.strip()) and ("`" * opener) not in inner
506
+
507
+
508
+ def prose_obligation_ranges(source: str) -> list[tuple[LedgerObligation, int, int]]:
509
+ """Recover ranges for prose obligations while excluding HTML comments.
510
+
511
+ Both walks are document ordered. Matching from the previous end makes a
512
+ repeated sentence under two different hosts addressable by the manifest's
513
+ (path, chain, exact text) identity instead of imposing corpus-wide textual
514
+ uniqueness.
515
+ """
516
+ visible = mask_inert_markdown(source)
517
+ haystack, offsets = normalized_with_offsets(visible)
518
+ cursor = 0
519
+ out: list[tuple[LedgerObligation, int, int]] = []
520
+ for parsed in GCD.parse(visible):
521
+ obligation = LedgerObligation(parsed.text, tuple(parsed.chain))
522
+ if inline_code_only(obligation.text):
523
+ continue
524
+ needle = GCD.normalize(obligation.text)
525
+ pos = haystack.find(needle, cursor)
526
+ if pos < 0:
527
+ raise AuditError(
528
+ "CARRIER_RANGE_UNRECOVERABLE",
529
+ f"cannot locate parsed obligation after normalized offset {cursor}: {needle[:80]}",
530
+ )
531
+ end_pos = pos + len(needle) - 1
532
+ out.append((obligation, offsets[pos], offsets[end_pos] + 1))
533
+ cursor = pos + len(needle)
534
+ return out
535
+
536
+
537
+ _TABLE_SEPARATOR_CELL = re.compile(r"^:?-{3,}:?$")
538
+
539
+
540
+ def code_span_mask(text: str) -> list[bool]:
541
+ """Mark characters inside matched inline code spans (CommonMark pairing).
542
+
543
+ A run of N backticks opens a span that closes only at the next run of
544
+ exactly N backticks; runs of a different length inside an open span are
545
+ literal, and an unmatched opener is literal text, not an open-forever
546
+ span. Backslash-escaped backticks outside spans are literal.
547
+ """
548
+ mask = [False] * len(text)
549
+ index = 0
550
+ while index < len(text):
551
+ char = text[index]
552
+ if char == "\\" and index + 1 < len(text):
553
+ index += 2
554
+ continue
555
+ if char != "`":
556
+ index += 1
557
+ continue
558
+ run = 1
559
+ while index + run < len(text) and text[index + run] == "`":
560
+ run += 1
561
+ search = index + run
562
+ closer = -1
563
+ while search < len(text):
564
+ if text[search] != "`":
565
+ search += 1
566
+ continue
567
+ closer_run = 1
568
+ while search + closer_run < len(text) and text[search + closer_run] == "`":
569
+ closer_run += 1
570
+ if closer_run == run:
571
+ closer = search
572
+ break
573
+ search += closer_run
574
+ if closer < 0:
575
+ index += run
576
+ continue
577
+ for position in range(index, closer + run):
578
+ mask[position] = True
579
+ index = closer + run
580
+ return mask
581
+
582
+
583
+ def table_cells(line: str) -> list[tuple[str, int, int]] | None:
584
+ """Split a pipe table without splitting escaped pipes or inline code."""
585
+ stripped = line.strip()
586
+ if not stripped.startswith("|") or not stripped.endswith("|"):
587
+ return None
588
+ leading = len(line) - len(line.lstrip())
589
+ content = stripped[1:-1]
590
+ in_code = code_span_mask(content)
591
+ cells: list[tuple[str, int, int]] = []
592
+
593
+ def emit(start: int, end: int) -> None:
594
+ raw_cell = content[start:end]
595
+ left_trim = len(raw_cell) - len(raw_cell.lstrip())
596
+ right_trimmed = raw_cell.rstrip()
597
+ cell_start = leading + 1 + start + left_trim
598
+ cell_end = leading + 1 + start + len(right_trimmed)
599
+ cells.append((GCD.normalize(raw_cell), cell_start, cell_end))
600
+
601
+ start = 0
602
+ escaped = False
603
+ for index, char in enumerate(content):
604
+ if escaped:
605
+ escaped = False
606
+ continue
607
+ if in_code[index]:
608
+ continue
609
+ if char == "\\":
610
+ escaped = True
611
+ continue
612
+ if char == "|":
613
+ emit(start, index)
614
+ start = index + 1
615
+ emit(start, len(content))
616
+ return cells
617
+
618
+
619
+ def is_table_separator(line: str) -> bool:
620
+ cells = table_cells(line)
621
+ return bool(cells) and all(
622
+ _TABLE_SEPARATOR_CELL.fullmatch(cell[0]) for cell in cells
623
+ )
624
+
625
+
626
+ def table_obligation_ranges(
627
+ source: str, *, include_inline_code_only: bool = False
628
+ ) -> list[tuple[LedgerObligation, int, int]]:
629
+ """Parse Markdown table data rows as obligations.
630
+
631
+ The header is table identity and therefore an immediate governing host.
632
+ Header/separator rows are schema, not obligations. Complete data-row text
633
+ preserves column/cell semantics and supplies an exact carrier locator.
634
+ """
635
+ out: list[tuple[LedgerObligation, int, int]] = []
636
+ headings: list[tuple[int, str]] = []
637
+ visible_source = mask_inert_markdown(source)
638
+ lines = visible_source.splitlines(keepends=True)
639
+ offsets: list[int] = []
640
+ total = 0
641
+ for line in lines:
642
+ offsets.append(total)
643
+ total += len(line)
644
+
645
+ table_header: list[str] | None = None
646
+ table_active = False
647
+ for index, raw_with_end in enumerate(lines):
648
+ visible = raw_with_end.rstrip("\r\n")
649
+
650
+ if not visible.strip():
651
+ table_header = None
652
+ table_active = False
653
+ continue
654
+
655
+ heading = GCD._HEADING.match(visible)
656
+ if heading:
657
+ level = len(heading.group(1))
658
+ while headings and headings[-1][0] >= level:
659
+ headings.pop()
660
+ headings.append((level, GCD.content_key(heading.group(2))))
661
+ table_header = None
662
+ table_active = False
663
+ continue
664
+
665
+ cells = table_cells(visible)
666
+ if cells is None:
667
+ table_header = None
668
+ table_active = False
669
+ continue
670
+ if is_table_separator(visible):
671
+ table_active = table_header is not None
672
+ continue
673
+ if not table_active:
674
+ table_header = [cell[0] for cell in cells]
675
+ continue
676
+
677
+ table_identity = " | ".join(table_header or [])
678
+ for column, (text, cell_start, cell_end) in enumerate(cells):
679
+ if not text or (inline_code_only(text) and not include_inline_code_only):
680
+ continue
681
+ header = (
682
+ table_header[column]
683
+ if table_header is not None and column < len(table_header)
684
+ else f"column-{column + 1}"
685
+ )
686
+ chain = tuple(key for _, key in headings) + (
687
+ "table:" + GCD.content_key(table_identity),
688
+ "column:" + GCD.content_key(header),
689
+ )
690
+ start = offsets[index] + cell_start
691
+ end = offsets[index] + cell_end
692
+ out.append((LedgerObligation(text, chain), start, end))
693
+ return out
694
+
695
+
696
+ def parse_obligation_ranges(source: str) -> list[tuple[LedgerObligation, int, int]]:
697
+ obligations = prose_obligation_ranges(source) + table_obligation_ranges(source)
698
+ return sorted(obligations, key=lambda item: (item[1], item[2]))
699
+
700
+
701
+ def raw_list_item_ranges(source: str) -> list[tuple[LedgerObligation, int, int]]:
702
+ """Expose an exact visible list item as a schema4 carrier without changing row derivation.
703
+
704
+ The legacy row splitter intentionally breaks label-plus-rule bullets into
705
+ clause-sized obligations. A schema4 bundle sometimes needs the complete
706
+ current item (label and operative clauses together). This locator is
707
+ limited to visible Markdown list items; comments, fences, and indented code
708
+ were already blanked by ``mask_inert_markdown``.
709
+ """
710
+ visible = mask_inert_markdown(source)
711
+ headings: list[tuple[int, str]] = []
712
+ list_stack: list[tuple[int, str]] = []
713
+ out: list[tuple[LedgerObligation, int, int]] = []
714
+ offset = 0
715
+ for raw in visible.splitlines(keepends=True):
716
+ line = raw.rstrip("\r\n")
717
+ heading = GCD._HEADING.match(line)
718
+ if heading:
719
+ level = len(heading.group(1))
720
+ while headings and headings[-1][0] >= level:
721
+ headings.pop()
722
+ headings.append((level, GCD.content_key(heading.group(2))))
723
+ list_stack.clear()
724
+ offset += len(raw)
725
+ continue
726
+ item = GCD._LIST_ITEM.match(line)
727
+ if item:
728
+ indent = len(item.group(1).expandtabs(4))
729
+ while list_stack and list_stack[-1][0] >= indent:
730
+ list_stack.pop()
731
+ text = GCD.normalize(item.group(2))
732
+ if text:
733
+ start = offset + item.start(2)
734
+ end = offset + item.end(2)
735
+ chain = tuple([value for _, value in headings] + [value for _, value in list_stack])
736
+ out.append((LedgerObligation(text, chain), start, end))
737
+ list_stack.append((indent, GCD.content_key(text)))
738
+ elif line and not line[0].isspace():
739
+ list_stack.clear()
740
+ offset += len(raw)
741
+ return out
742
+
743
+
744
+ def raw_table_row_ranges(source: str) -> list[tuple[LedgerObligation, int, int]]:
745
+ """Expose a complete visible data row, never an inline-code-only cell."""
746
+ visible = mask_inert_markdown(source)
747
+ headings: list[tuple[int, str]] = []
748
+ header: list[str] | None = None
749
+ active = False
750
+ out: list[tuple[LedgerObligation, int, int]] = []
751
+ offset = 0
752
+ for raw in visible.splitlines(keepends=True):
753
+ line = raw.rstrip("\r\n")
754
+ heading = GCD._HEADING.match(line)
755
+ if heading:
756
+ level = len(heading.group(1))
757
+ while headings and headings[-1][0] >= level:
758
+ headings.pop()
759
+ headings.append((level, GCD.content_key(heading.group(2))))
760
+ header = None
761
+ active = False
762
+ offset += len(raw)
763
+ continue
764
+ cells = table_cells(line)
765
+ if cells is None:
766
+ header = None
767
+ active = False
768
+ offset += len(raw)
769
+ continue
770
+ if is_table_separator(line):
771
+ active = header is not None
772
+ offset += len(raw)
773
+ continue
774
+ if not active:
775
+ header = [cell[0] for cell in cells]
776
+ offset += len(raw)
777
+ continue
778
+ text = " | ".join(cell[0] for cell in cells)
779
+ if text:
780
+ identity = " | ".join(header or [])
781
+ chain = tuple(value for _, value in headings) + (
782
+ "table:" + GCD.content_key(identity),
783
+ "row",
784
+ )
785
+ out.append(
786
+ (
787
+ LedgerObligation(text, chain),
788
+ offset + cells[0][1],
789
+ offset + cells[-1][2],
790
+ )
791
+ )
792
+ offset += len(raw)
793
+ return out
794
+
795
+
796
+ def section_body_ranges(source: str) -> list[tuple[LedgerObligation, int, int]]:
797
+ """Return exact visible section bodies for schema4 closed-bundle carriers."""
798
+ visible = mask_inert_markdown(source)
799
+ lines = visible.splitlines(keepends=True)
800
+ offsets: list[int] = []
801
+ total = 0
802
+ for line in lines:
803
+ offsets.append(total)
804
+ total += len(line)
805
+ headings: list[tuple[int, str]] = []
806
+ found: list[tuple[int, int, tuple[str, ...]]] = []
807
+ for index, raw in enumerate(lines):
808
+ heading = GCD._HEADING.match(raw.rstrip("\r\n"))
809
+ if not heading:
810
+ continue
811
+ level = len(heading.group(1))
812
+ while headings and headings[-1][0] >= level:
813
+ headings.pop()
814
+ chain = tuple([value for _, value in headings] + [GCD.content_key(heading.group(2))])
815
+ found.append((index, level, chain))
816
+ headings.append((level, GCD.content_key(heading.group(2))))
817
+ out: list[tuple[LedgerObligation, int, int]] = []
818
+ for position, (line_index, level, chain) in enumerate(found):
819
+ next_index = len(lines)
820
+ for candidate_index, candidate_level, _ in found[position + 1 :]:
821
+ if candidate_level <= level:
822
+ next_index = candidate_index
823
+ break
824
+ start = offsets[line_index] + len(lines[line_index])
825
+ end = offsets[next_index] if next_index < len(lines) else len(visible)
826
+ while start < end and visible[start].isspace():
827
+ start += 1
828
+ while end > start and visible[end - 1].isspace():
829
+ end -= 1
830
+ text = GCD.normalize(visible[start:end])
831
+ if text:
832
+ out.append((LedgerObligation(text, chain), start, end))
833
+ return out
834
+
835
+
836
+ def exact_carrier(repo: Path, path: str, text: str, chain: list[str]) -> Carrier:
837
+ file_path = repo / path
838
+ if not file_path.is_file():
839
+ raise AuditError("CARRIER_PATH_MISSING", path)
840
+ source = file_path.read_text(encoding="utf-8")
841
+ normalized_text = GCD.normalize(text)
842
+ claimed_chain = tuple(chain)
843
+ exact_text_obligations = [
844
+ (obligation, start, end)
845
+ for obligation, start, end in parse_obligation_ranges(source)
846
+ if GCD.normalize(obligation.text) == normalized_text
847
+ ]
848
+ if not exact_text_obligations:
849
+ exact_text_obligations = [
850
+ (obligation, start, end)
851
+ for obligation, start, end in (
852
+ raw_list_item_ranges(source)
853
+ + raw_table_row_ranges(source)
854
+ + section_body_ranges(source)
855
+ )
856
+ if GCD.normalize(obligation.text) == normalized_text
857
+ ]
858
+ matching_obligations = [
859
+ item for item in exact_text_obligations if tuple(item[0].chain) == claimed_chain
860
+ ]
861
+ if not matching_obligations and exact_text_obligations:
862
+ raise AuditError(
863
+ "CARRIER_CHAIN_MISMATCH",
864
+ f"{path}: exact text exists, but not under claimed chain {list(claimed_chain)!r}",
865
+ )
866
+ if len(matching_obligations) != 1:
867
+ raise AuditError(
868
+ "CARRIER_COMPOSITE_NOT_UNIQUE",
869
+ f"{path}: (chain, exact text) count={len(matching_obligations)}",
870
+ )
871
+ obligation, start, end = matching_obligations[0]
872
+ derived_chain = tuple(obligation.chain)
873
+ start_line, end_line = line_range(source, start, end)
874
+ return Carrier(path, normalized_text, derived_chain, start_line, end_line)
875
+
876
+
877
+ QUALIFIER_PATTERNS: dict[str, re.Pattern[str]] = {
878
+ "modality": re.compile(
879
+ r"\b(?:must|required|shall|never|cannot|can't|do not|may|should)\b|必须|不得|禁止|不可|不能|应当|应该|可以",
880
+ re.I,
881
+ ),
882
+ "recency": re.compile(
883
+ r"\b(?:before|after|current|latest|final|again|first|initial|prior|subsequent)\b|之前|之后|当前|最新|最终|再次|首次|先|再",
884
+ re.I,
885
+ ),
886
+ "scope": re.compile(
887
+ r"\b(?:not\s+a\s+substitute|limited\s+to|restricted\s+to|every|each|all|any|only|solely|exclusively|full|whole|per|none)\b|每个|每条|全部|所有|任何|仅|完整|全量|无一",
888
+ re.I,
889
+ ),
890
+ "actor": re.compile(
891
+ r"\b(?:owners?|users?|reviewers?|testers?|testing|clients?|maintainers?|designers?|developers?|agents?)\b|用户|负责人|维护者|评审者|设计师|开发者|测试人员|测试者|客户端|智能体",
892
+ re.I,
893
+ ),
894
+ "consequence": re.compile(
895
+ r"\b(?:blocks?|blocked|fails?|failed|stops?|stopped|invalid|rejects?|rejected|pending|cannot claim|must not claim|red)\b|阻塞|失败|停止|无效|拒绝|待定|不得声称|不可声明|红灯",
896
+ re.I,
897
+ ),
898
+ }
899
+
900
+ HARD_MODALITY = re.compile(
901
+ r"\b(?:must|required|shall|never|cannot|can't|do not)\b|必须|不得|禁止|不可|不能",
902
+ re.I,
903
+ )
904
+ SOFT_MODALITY = re.compile(r"\b(?:may|should|can)\b|可以|应该|应当", re.I)
905
+
906
+ _THRESHOLD_COMPARATOR = re.compile(
907
+ r"\b(?:at least|at most|minimum|maximum|more than|less than)\b"
908
+ r"|至少|至多|最多|最少|不超过|不少于|高于|低于",
909
+ re.I,
910
+ )
911
+ _THRESHOLD_DIRECTIONAL_COMPARATOR = re.compile(
912
+ r"\b(?:above|below|over|under)\b(?=\s+\d+(?:\.\d+)?)",
913
+ re.I,
914
+ )
915
+ _THRESHOLD_WITHIN = re.compile(
916
+ r"\bwithin\b(?=\s+(?:\d+(?:\.\d+)?|one|two|three|four|five|six|seven|eight|nine|ten)\s*(?:ms|msec(?:ond)?s?|s|sec(?:ond)?s?|min(?:ute)?s?|h|hours?|days?|frames?|attempts?|retries?|requests?)\b)",
917
+ re.I,
918
+ )
919
+ _THRESHOLD_RATIO = re.compile(
920
+ r"(?<![A-Za-z0-9_])\d+(?:\.\d+)?\s*:\s*\d+(?:\.\d+)?(?![A-Za-z0-9_])"
921
+ )
922
+ _THRESHOLD_DIMENSION = re.compile(
923
+ r"(?<![A-Za-z0-9_])"
924
+ r"\d+(?:\.\d+)?\s*[x×]\s*\d+(?:\.\d+)?"
925
+ r"(?:\s*(?:px|pt|dp|sp))?"
926
+ r"(?![A-Za-z0-9_])",
927
+ re.I,
928
+ )
929
+ _THRESHOLD_QUANTITY = re.compile(
930
+ r"(?<![A-Za-z0-9_])"
931
+ r"\d+(?:\.\d+)?\s*"
932
+ r"(?:%|px|pt|dp|sp|ms|msec(?:ond)?s?|s|sec(?:ond)?s?|min(?:ute)?s?|h|hours?|days?|bytes?|kib|mib|gib|kb|mb|gb|items?|rows?|files?|steps?|times?|characters?|chars?|screens?|locales?|variants?|states?|frames?|attempts?|retries?|requests?|users?|个|项|条|次|秒|分钟|小时|天|像素|字符|行|列|页|屏|帧|毫秒)"
933
+ r"(?![A-Za-z0-9_])",
934
+ re.I,
935
+ )
936
+ _ACTOR_MODIFIER_AFTER = re.compile(
937
+ r"^(?:[-‑](?:facing|visible|authored|generated|provided)\b"
938
+ r"|\s+(?:copy|language|interface|research)\b"
939
+ r"|\s+experience\s+research\b"
940
+ r"|(?:['’]s?|s['’])\s+needs\b"
941
+ r"|\s+needs\b(?!\s+to\b))",
942
+ re.I,
943
+ )
944
+ _CHINESE_USER_NON_ACTOR_AFTER = re.compile(
945
+ r"^(?:的?需求|文案|语言|界面|可见|知道|看到|研究)"
946
+ )
947
+ _BLOCK_MODAL_BEFORE = re.compile(
948
+ r"(?:\b(?:must|shall|should|may|can|will|would|to)\b(?:\s+[A-Za-z-]+){0,3}|(?:必须|应当|应该|可以|可)(?:\S{0,6}))\s*$",
949
+ re.I,
950
+ )
951
+ _FINAL_TERMINAL_NOUN = re.compile(
952
+ r"^\s+(?:approval|acceptance|artifact|checkpoint|response|verdict|submission|state|content|screen|result|release|merge|commit|push|ready|completion)\b",
953
+ re.I,
954
+ )
955
+ _CLOSED_FINAL_RELATION_SOURCE = re.compile(
956
+ r"^(?:the\s+)?final\s+skill\s+(?:must|shall)\s+(?:stay|remain)\s+generic[.!]?$",
957
+ re.I,
958
+ )
959
+
960
+ POLARITY_GROUPS: dict[str, list[tuple[re.Pattern[str], re.Pattern[str]]]] = {
961
+ "recency": [
962
+ (
963
+ re.compile(r"\b(?:before|prior)\b|之前|先", re.I),
964
+ re.compile(r"\b(?:after|subsequent)\b|之后|再", re.I),
965
+ ),
966
+ (
967
+ re.compile(r"\b(?:latest|final)\b|最新|最终", re.I),
968
+ re.compile(r"\b(?:first|initial)\b|首次", re.I),
969
+ ),
970
+ ],
971
+ "threshold": [
972
+ (
973
+ re.compile(
974
+ r"\b(?:at least|minimum|more than)\b|\b(?:above|over)\b(?=\s+\d)|至少|最少|不少于|高于",
975
+ re.I,
976
+ ),
977
+ re.compile(
978
+ r"\b(?:at most|maximum|less than)\b|\b(?:below|under)\b(?=\s+\d)|至多|最多|不超过|低于",
979
+ re.I,
980
+ ),
981
+ )
982
+ ],
983
+ "scope": [
984
+ (
985
+ re.compile(
986
+ r"\b(?:not\s+a\s+substitute|limited\s+to|restricted\s+to|every|each|all|only|solely|exclusively|none)\b|每个|每条|全部|所有|仅|无一",
987
+ re.I,
988
+ ),
989
+ re.compile(r"\bany\b|任何", re.I),
990
+ )
991
+ ],
992
+ }
993
+
994
+
995
+ def _dedupe_terms(matches: Iterable[str]) -> list[str]:
996
+ seen: set[str] = set()
997
+ values: list[str] = []
998
+ for value in matches:
999
+ if value in seen:
1000
+ continue
1001
+ seen.add(value)
1002
+ values.append(value)
1003
+ return values
1004
+
1005
+
1006
+ _QUOTED_LITERAL = re.compile(
1007
+ r'"(?:\\.|[^"\\])*"'
1008
+ r"|(?<![A-Za-z0-9])'(?:\\.|[^'\\])*'(?![A-Za-z0-9])"
1009
+ r"|“[^”]*”|‘[^’]*’"
1010
+ )
1011
+ _PATH_LITERAL = re.compile(
1012
+ r"(?<![A-Za-z0-9_./-])"
1013
+ r"(?:(?:\.{0,2}/|/)[^\s`\"']+|[^\s`\"']+/[^\s`\"']+\.[A-Za-z0-9]{1,8})"
1014
+ )
1015
+
1016
+
1017
+ def _mask_threshold_literals(text: str) -> str:
1018
+ """Blank inline code, quoted literals, and path tokens at stable offsets."""
1019
+ chars = list(text)
1020
+
1021
+ cursor = 0
1022
+ while cursor < len(text):
1023
+ start = text.find("`", cursor)
1024
+ if start < 0:
1025
+ break
1026
+ run = 1
1027
+ while start + run < len(text) and text[start + run] == "`":
1028
+ run += 1
1029
+ marker = "`" * run
1030
+ end = text.find(marker, start + run)
1031
+ if end < 0:
1032
+ cursor = start + run
1033
+ continue
1034
+ _blank(chars, start, end + run)
1035
+ cursor = end + run
1036
+
1037
+ for pattern in (_QUOTED_LITERAL, _PATH_LITERAL):
1038
+ visible = "".join(chars)
1039
+ for match in pattern.finditer(visible):
1040
+ _blank(chars, match.start(), match.end())
1041
+ return "".join(chars)
1042
+
1043
+
1044
+ def _threshold_terms(text: str) -> list[str]:
1045
+ searchable = _mask_threshold_literals(text)
1046
+ matches: list[tuple[int, int, str]] = []
1047
+ occupied: list[tuple[int, int]] = []
1048
+ for pattern in (_THRESHOLD_DIMENSION, _THRESHOLD_RATIO, _THRESHOLD_QUANTITY):
1049
+ for match in pattern.finditer(searchable):
1050
+ if any(match.start() < end and start < match.end() for start, end in occupied):
1051
+ continue
1052
+ occupied.append((match.start(), match.end()))
1053
+ matches.append((match.start(), match.end(), match.group(0)))
1054
+ for pattern in (
1055
+ _THRESHOLD_COMPARATOR,
1056
+ _THRESHOLD_DIRECTIONAL_COMPARATOR,
1057
+ _THRESHOLD_WITHIN,
1058
+ ):
1059
+ for match in pattern.finditer(searchable):
1060
+ matches.append((match.start(), match.end(), match.group(0)))
1061
+ matches.sort(key=lambda item: (item[0], item[1]))
1062
+ return _dedupe_terms(value for _, _, value in matches)
1063
+
1064
+
1065
+ def _recency_terms(text: str) -> list[str]:
1066
+ values = []
1067
+ for match in QUALIFIER_PATTERNS["recency"].finditer(text):
1068
+ if match.group(0).casefold() == "first" and re.match(
1069
+ r"[-‑]screen\b", text[match.end() :], re.I
1070
+ ):
1071
+ continue
1072
+ values.append(match.group(0))
1073
+ return _dedupe_terms(values)
1074
+
1075
+
1076
+ def _scope_terms(text: str) -> list[str]:
1077
+ values = []
1078
+ for match in QUALIFIER_PATTERNS["scope"].finditer(text):
1079
+ term = match.group(0)
1080
+ before = text[: match.start()]
1081
+ if term.casefold() == "only" and re.search(
1082
+ r"\bnot\s+(?:[*_~]{1,3})?$", before, re.I
1083
+ ):
1084
+ continue
1085
+ if term == "仅" and re.search(r"不(?:仅)?$", before):
1086
+ continue
1087
+ if term.casefold() == "per" and re.match(
1088
+ r"\s+(?:[*_~]{1,3})?the\b", text[match.end() :], re.I
1089
+ ):
1090
+ # "per the classes/design above" is an according-to cross-reference,
1091
+ # not a closed-set quantifier such as "per affected client".
1092
+ continue
1093
+ values.append(term)
1094
+ return _dedupe_terms(values)
1095
+
1096
+
1097
+ def _actor_terms(text: str) -> list[str]:
1098
+ values = []
1099
+ for match in QUALIFIER_PATTERNS["actor"].finditer(text):
1100
+ term = match.group(0)
1101
+ if term.casefold() in {"user", "users"}:
1102
+ if _ACTOR_MODIFIER_AFTER.match(text[match.end() :]):
1103
+ continue
1104
+ elif term == "用户":
1105
+ if _CHINESE_USER_NON_ACTOR_AFTER.match(text[match.end() :]):
1106
+ continue
1107
+ elif term.casefold() == "agent" and re.match(
1108
+ r"[-‑](?:contract|facing|generated|owned)\b",
1109
+ text[match.end() :],
1110
+ re.I,
1111
+ ):
1112
+ continue
1113
+ elif term.casefold() == "reviewer" and re.match(
1114
+ r"\s+(?:checks?|checklist|criteria|检查项)\b",
1115
+ text[match.end() :],
1116
+ re.I,
1117
+ ):
1118
+ continue
1119
+ values.append(term)
1120
+ return _dedupe_terms(values)
1121
+
1122
+
1123
+ def _inside_code_path(text: str, start: int, end: int) -> bool:
1124
+ if text[:start].count("`") % 2 == 0:
1125
+ return False
1126
+ left = text.rfind("`", 0, start)
1127
+ right = text.find("`", end)
1128
+ if left < 0 or right < 0:
1129
+ return False
1130
+ token = text[left + 1 : right]
1131
+ return "/" in token or re.search(r"(?:^|[-_.])(?:md|txt|json|ya?ml|py|sh)$", token, re.I) is not None
1132
+
1133
+
1134
+ def _block_is_enumerated_noun(text: str, start: int, end: int) -> bool:
1135
+ before = text[max(0, start - 48) : start]
1136
+ after = text[end : end + 48]
1137
+ if _BLOCK_MODAL_BEFORE.search(before):
1138
+ return False
1139
+ previous = before.rstrip()[-1:] if before.rstrip() else ""
1140
+ following = after.lstrip()
1141
+ return previous in {",", "/"} and bool(
1142
+ re.match(r"^(?:,|/|\band\b|\bor\b|,|、|和|或)", following, re.I)
1143
+ )
1144
+
1145
+
1146
+ def _block_is_code_noun(text: str, start: int) -> bool:
1147
+ return re.search(r"\bcode\s+$", text[max(0, start - 24) : start], re.I) is not None
1148
+
1149
+
1150
+ def _red_is_color_term(text: str, start: int, end: int) -> bool:
1151
+ before = text[max(0, start - 32) : start]
1152
+ after = text[end : end + 32]
1153
+ return bool(
1154
+ re.search(r"\b(?:color|colour)(?:\s+is|\s*:)?\s*$", before, re.I)
1155
+ or re.match(r"^[-‑]colou?red\b", after, re.I)
1156
+ or re.match(r"^(?:[-‑]|\s+)(?:first|baseline)\b", after, re.I)
1157
+ or re.match(
1158
+ r"^\s+(?:color|colour|text|border|background|icon|badge|token|fill|stroke)\b",
1159
+ after,
1160
+ re.I,
1161
+ )
1162
+ )
1163
+
1164
+
1165
+ def _consequence_terms(text: str) -> list[str]:
1166
+ values = []
1167
+ for match in QUALIFIER_PATTERNS["consequence"].finditer(text):
1168
+ term = match.group(0)
1169
+ folded = term.casefold()
1170
+ if _inside_code_path(text, match.start(), match.end()):
1171
+ continue
1172
+ if folded == "block" and _block_is_enumerated_noun(
1173
+ text, match.start(), match.end()
1174
+ ):
1175
+ continue
1176
+ if folded in {"block", "blocks"} and _block_is_code_noun(
1177
+ text, match.start()
1178
+ ):
1179
+ continue
1180
+ if folded == "fail" and (
1181
+ text[match.start() - 1 : match.start()] == "/"
1182
+ or text[match.end() : match.end() + 1] == "/"
1183
+ ):
1184
+ continue
1185
+ if folded == "red" and _red_is_color_term(
1186
+ text, match.start(), match.end()
1187
+ ):
1188
+ continue
1189
+ values.append(term)
1190
+ return _dedupe_terms(values)
1191
+
1192
+
1193
+ def qualifier_terms(text: str) -> dict[str, list[str]]:
1194
+ extractors = {
1195
+ "modality": lambda value: _dedupe_terms(
1196
+ match.group(0) for match in QUALIFIER_PATTERNS["modality"].finditer(value)
1197
+ ),
1198
+ "recency": _recency_terms,
1199
+ "threshold": _threshold_terms,
1200
+ "scope": _scope_terms,
1201
+ "actor": _actor_terms,
1202
+ "consequence": _consequence_terms,
1203
+ }
1204
+ result: dict[str, list[str]] = {}
1205
+ for kind in ("modality", "recency", "threshold", "scope", "actor", "consequence"):
1206
+ terms = extractors[kind](text)
1207
+ if terms:
1208
+ result[kind] = terms
1209
+ return result
1210
+
1211
+
1212
+ _QUALIFIER_RELATION_FIELDS = {
1213
+ "kind",
1214
+ "term",
1215
+ "occurrence",
1216
+ "source_excerpt",
1217
+ "resolution",
1218
+ "same_immediate_host",
1219
+ }
1220
+ _DIRECTIONAL_RECENCY_TERMS = {
1221
+ "before",
1222
+ "after",
1223
+ "prior",
1224
+ "subsequent",
1225
+ "之前",
1226
+ "之后",
1227
+ "先",
1228
+ "再",
1229
+ }
1230
+
1231
+
1232
+ def _literal_occurrences(text: str, term: str) -> list[re.Match[str]]:
1233
+ return list(
1234
+ re.finditer(
1235
+ rf"(?<![A-Za-z0-9_]){re.escape(term)}(?![A-Za-z0-9_])",
1236
+ text,
1237
+ re.I,
1238
+ )
1239
+ )
1240
+
1241
+
1242
+ def _is_closed_non_temporal_final_relation(
1243
+ source_text: str, carrier_text: str, occurrence: re.Match[str]
1244
+ ) -> bool:
1245
+ normalized_source = GCD.normalize(source_text)
1246
+ if not _CLOSED_FINAL_RELATION_SOURCE.fullmatch(normalized_source):
1247
+ return False
1248
+ rewritten = (
1249
+ normalized_source[: occurrence.start()]
1250
+ + "resulting"
1251
+ + normalized_source[occurrence.end() :]
1252
+ )
1253
+ return GCD.normalize(rewritten).casefold() == GCD.normalize(carrier_text).casefold()
1254
+
1255
+
1256
+ def validate_qualifier_relations(
1257
+ expected: ExpectedRow,
1258
+ mapping: dict[str, Any],
1259
+ carrier: Carrier,
1260
+ required: dict[str, list[str]],
1261
+ ) -> dict[str, list[str]]:
1262
+ label = f"{expected.source_path}#{expected.source_ordinal}"
1263
+ relations = mapping.get("qualifier_relations", [])
1264
+ if not isinstance(relations, list):
1265
+ raise AuditError("QUALIFIER_RELATION_INVALID", f"{label}: array required")
1266
+ if not relations:
1267
+ return required
1268
+ if mapping.get("effect") != "strengthened":
1269
+ raise AuditError("QUALIFIER_RELATION_REQUIRES_STRENGTHENED", label)
1270
+ if mapping.get("manual_reviewed") is not True:
1271
+ raise AuditError("STRENGTHENED_REVIEW_REQUIRED", label)
1272
+
1273
+ effective = {kind: list(terms) for kind, terms in required.items()}
1274
+ seen: set[tuple[str, str, int]] = set()
1275
+ for relation in relations:
1276
+ if not isinstance(relation, dict) or set(relation) != _QUALIFIER_RELATION_FIELDS:
1277
+ raise AuditError("QUALIFIER_RELATION_INVALID", label)
1278
+ kind = relation.get("kind")
1279
+ if kind != "recency":
1280
+ raise AuditError(
1281
+ "QUALIFIER_RELATION_FORBIDDEN", f"{label}: kind={kind}"
1282
+ )
1283
+ if relation.get("resolution") != "carrier-applies-unconditionally":
1284
+ raise AuditError("QUALIFIER_RELATION_INVALID", f"{label}: resolution")
1285
+ if relation.get("same_immediate_host") is not True:
1286
+ raise AuditError("QUALIFIER_RELATION_INVALID", f"{label}: host")
1287
+ term = relation.get("term")
1288
+ occurrence = relation.get("occurrence")
1289
+ excerpt = relation.get("source_excerpt")
1290
+ if (
1291
+ not isinstance(term, str)
1292
+ or not term
1293
+ or not isinstance(occurrence, int)
1294
+ or isinstance(occurrence, bool)
1295
+ or occurrence < 1
1296
+ or not isinstance(excerpt, str)
1297
+ or not excerpt
1298
+ or "\n" in excerpt
1299
+ or len(excerpt) > 120
1300
+ ):
1301
+ raise AuditError("QUALIFIER_RELATION_INVALID", label)
1302
+ identity = (str(kind), term.casefold(), occurrence)
1303
+ if identity in seen:
1304
+ raise AuditError("QUALIFIER_RELATION_INVALID", f"{label}: duplicate")
1305
+ seen.add(identity)
1306
+
1307
+ source_terms = effective.get("recency", [])
1308
+ matching_source_terms = [
1309
+ value for value in source_terms if value.casefold() == term.casefold()
1310
+ ]
1311
+ occurrences = _literal_occurrences(expected.before_text, term)
1312
+ excerpt_start = expected.before_text.find(excerpt)
1313
+ if (
1314
+ len(matching_source_terms) != 1
1315
+ or len(occurrences) != 1
1316
+ or occurrence != 1
1317
+ or expected.before_text.count(excerpt) != 1
1318
+ or excerpt_start < 0
1319
+ or not (
1320
+ excerpt_start <= occurrences[0].start()
1321
+ and occurrences[0].end() <= excerpt_start + len(excerpt)
1322
+ )
1323
+ ):
1324
+ raise AuditError("QUALIFIER_RELATION_INVALID", label)
1325
+
1326
+ other_recency = [
1327
+ value for value in source_terms if value.casefold() != term.casefold()
1328
+ ]
1329
+ source_consequences = required.get("consequence", [])
1330
+ after_recency = qualifier_terms(carrier.text).get("recency", [])
1331
+ terminal_tail = expected.before_text[occurrences[0].end() :]
1332
+ if (
1333
+ term.casefold() != "final"
1334
+ or not _is_closed_non_temporal_final_relation(
1335
+ expected.before_text, carrier.text, occurrences[0]
1336
+ )
1337
+ or term.casefold() in _DIRECTIONAL_RECENCY_TERMS
1338
+ or other_recency
1339
+ or source_consequences
1340
+ or _FINAL_TERMINAL_NOUN.match(terminal_tail)
1341
+ or after_recency
1342
+ ):
1343
+ raise AuditError("QUALIFIER_RELATION_FORBIDDEN", label)
1344
+
1345
+ effective["recency"] = [
1346
+ value for value in source_terms if value.casefold() != term.casefold()
1347
+ ]
1348
+ if not effective["recency"]:
1349
+ del effective["recency"]
1350
+ return effective
1351
+
1352
+
1353
+ _CLAUSE_ACTION = re.compile(
1354
+ r"\b(?:must|shall|required|record|verify|preserve|reject|block|expose|show|keep|route|check|name|include|cover|provide|ensure|prevent|allow|forbid)\b|必须|应当|记录|验证|保留|拒绝|阻塞|展示|路由|检查|命名|包括|覆盖|提供|确保|防止|允许|禁止",
1355
+ re.I,
1356
+ )
1357
+ _CLAUSE_LOCAL_ACTION = re.compile(
1358
+ r"\b(?:record|verify|preserve|reject|block|expose|show|keep|route|check|name|include|cover|provide|ensure|prevent|allow|forbid|retain|confirm|complete|finish|remain|stay)\b|记录|验证|保留|拒绝|阻塞|展示|路由|检查|命名|包括|覆盖|提供|确保|防止|允许|禁止|保有|确认|完成|保持|测试",
1359
+ re.I,
1360
+ )
1361
+
1362
+
1363
+ def _mask_code_spans_text(text: str) -> str:
1364
+ """Same-length copy with inline code-span characters replaced by NUL.
1365
+
1366
+ Clause delimiters (semicolons, colons, commas, join words) inside inline
1367
+ code are literal content, not clause structure, so structural scans run
1368
+ on this masked copy while clause text is sliced from the original.
1369
+ """
1370
+ mask = code_span_mask(text)
1371
+ return "".join(
1372
+ "\x00" if in_code else char for char, in_code in zip(text, mask)
1373
+ )
1374
+
1375
+
1376
+ def compound_clauses(text: str) -> list[dict[str, str]]:
1377
+ """Conservatively and reproducibly split a baseline compound obligation."""
1378
+ normalized = GCD.normalize(text)
1379
+ structural = _mask_code_spans_text(normalized)
1380
+
1381
+ def split_outside_code(value: str, value_structural: str, pattern: str) -> list[str]:
1382
+ parts: list[str] = []
1383
+ start = 0
1384
+ for match in re.finditer(pattern, value_structural):
1385
+ parts.append(value[start : match.start()])
1386
+ start = match.end()
1387
+ parts.append(value[start:])
1388
+ return parts
1389
+
1390
+ semicolon_parts = [
1391
+ GCD.normalize(part)
1392
+ for part in split_outside_code(normalized, structural, r"[;;]")
1393
+ ]
1394
+ semicolon_parts = [part for part in semicolon_parts if part]
1395
+ if len(semicolon_parts) >= 2:
1396
+ return [
1397
+ {"id": f"c{index}", "text": part}
1398
+ for index, part in enumerate(semicolon_parts, 1)
1399
+ ]
1400
+
1401
+ colon = re.match(r"^(.*?[::])\s*(.+)$", structural)
1402
+ if not colon or not _CLAUSE_ACTION.search(structural[: colon.end(1)]):
1403
+ return []
1404
+ tail = normalized[colon.start(2) :]
1405
+ tail_structural = structural[colon.start(2) :]
1406
+ final_join = re.search(
1407
+ r"(?:,|,)\s*(?:and|or|以及|与|或)\s+", tail_structural, re.I
1408
+ )
1409
+ if not final_join:
1410
+ return []
1411
+ expanded = tail[: final_join.start()] + ", " + tail[final_join.end() :]
1412
+ expanded_structural = (
1413
+ tail_structural[: final_join.start()]
1414
+ + ", "
1415
+ + tail_structural[final_join.end() :]
1416
+ )
1417
+ item_pairs = [
1418
+ (GCD.normalize(item), item_structural)
1419
+ for item, item_structural in zip(
1420
+ split_outside_code(expanded, expanded_structural, r"[,,]"),
1421
+ split_outside_code(expanded_structural, expanded_structural, r"[,,]"),
1422
+ )
1423
+ ]
1424
+ item_pairs = [pair for pair in item_pairs if pair[0]]
1425
+ items = [item for item, _ in item_pairs]
1426
+ if len(items) < 3:
1427
+ return []
1428
+ clause_like = sum(
1429
+ bool(_CLAUSE_ACTION.search(item_structural))
1430
+ or GCD.effective_len(item) >= 24
1431
+ for item, item_structural in item_pairs
1432
+ )
1433
+ if clause_like < 2:
1434
+ return []
1435
+ parts = [normalized[: colon.end(1)] + " " + items[0], *items[1:]]
1436
+ return [
1437
+ {"id": f"c{index}", "text": part}
1438
+ for index, part in enumerate(parts, 1)
1439
+ ]
1440
+
1441
+
1442
+ def ensure_qualifier_strength(before: str, after: str, label: str) -> None:
1443
+ required = qualifier_terms(before)
1444
+ available = qualifier_terms(after)
1445
+ missing = sorted(set(required) - set(available))
1446
+ if missing:
1447
+ raise AuditError(
1448
+ "BUNDLE_QUALIFIER_MISSING", f"{label}: classes={','.join(missing)}"
1449
+ )
1450
+ if HARD_MODALITY.search(before) and not HARD_MODALITY.search(after):
1451
+ raise AuditError("BUNDLE_QUALIFIER_WEAKENED", label)
1452
+ for polarity_kind in required:
1453
+ for positive, negative in POLARITY_GROUPS.get(polarity_kind, []):
1454
+ if positive.search(before) and negative.search(after):
1455
+ raise AuditError("BUNDLE_QUALIFIER_REVERSED", f"{label}: {polarity_kind}")
1456
+ if negative.search(before) and positive.search(after):
1457
+ raise AuditError("BUNDLE_QUALIFIER_REVERSED", f"{label}: {polarity_kind}")
1458
+ for kind, before_terms in required.items():
1459
+ before_literals = {term.casefold() for term in before_terms}
1460
+ after_literals = {term.casefold() for term in available.get(kind, [])}
1461
+ absent = sorted(before_literals - after_literals)
1462
+ if absent:
1463
+ raise AuditError(
1464
+ "BUNDLE_QUALIFIER_LITERAL_MISSING",
1465
+ f"{label}: {kind}={','.join(absent)}",
1466
+ )
1467
+ for kind in required:
1468
+ for positive, negative in POLARITY_GROUPS.get(kind, []):
1469
+ if positive.search(before) and negative.search(after):
1470
+ raise AuditError("BUNDLE_QUALIFIER_REVERSED", f"{label}: {kind}")
1471
+ if negative.search(before) and positive.search(after):
1472
+ raise AuditError("BUNDLE_QUALIFIER_REVERSED", f"{label}: {kind}")
1473
+
1474
+
1475
+ def _clause_action_terms(text: str) -> list[str]:
1476
+ return _dedupe_terms(
1477
+ match.group(0) for match in _CLAUSE_LOCAL_ACTION.finditer(text)
1478
+ )
1479
+
1480
+
1481
+ def _missing_terms(required: Iterable[str], available: Iterable[str]) -> list[str]:
1482
+ wanted = Counter(value.casefold() for value in required)
1483
+ present = Counter(value.casefold() for value in available)
1484
+ return sorted((wanted - present).elements())
1485
+
1486
+
1487
+ def validate_semantic_review(
1488
+ evidence: dict[str, Any], label: str, *, required: bool
1489
+ ) -> None:
1490
+ """Validate review-evidence shape, never the truth of the review decision."""
1491
+ status = evidence.get("semantic_review")
1492
+ rationale = evidence.get("semantic_rationale")
1493
+ if required:
1494
+ if status != "reviewed":
1495
+ raise AuditError("SEMANTIC_REVIEW_REQUIRED", label)
1496
+ if not isinstance(rationale, str) or not rationale.strip():
1497
+ raise AuditError("SEMANTIC_RATIONALE_REQUIRED", label)
1498
+ return
1499
+ if status is not None or rationale is not None:
1500
+ raise AuditError("SEMANTIC_REVIEW_REDUNDANT", label)
1501
+
1502
+
1503
+ def ensure_clause_local_lexical_guard(before: str, after: str, label: str) -> None:
1504
+ """Catch literal local omissions without claiming semantic equivalence."""
1505
+ before_qualifiers = qualifier_terms(before)
1506
+ after_qualifiers = qualifier_terms(after)
1507
+ for kind in ("modality", "actor", "consequence"):
1508
+ missing = _missing_terms(
1509
+ before_qualifiers.get(kind, []), after_qualifiers.get(kind, [])
1510
+ )
1511
+ if missing:
1512
+ raise AuditError(
1513
+ f"BUNDLE_CLAUSE_{kind.upper()}_MISSING",
1514
+ f"{label}: {','.join(missing)}",
1515
+ )
1516
+ missing_actions = _missing_terms(
1517
+ _clause_action_terms(before), _clause_action_terms(after)
1518
+ )
1519
+ if missing_actions:
1520
+ raise AuditError(
1521
+ "BUNDLE_CLAUSE_ACTION_MISSING",
1522
+ f"{label}: {','.join(missing_actions)}",
1523
+ )
1524
+
1525
+
1526
+ def validate_bundle_clause_proof(
1527
+ before: str,
1528
+ after: str,
1529
+ claimed: Any,
1530
+ label: str,
1531
+ ) -> None:
1532
+ # Open-text actor/action/consequence equivalence is review-owned. These
1533
+ # deterministic checks are lexical tripwires, not semantic proof.
1534
+ ensure_clause_local_lexical_guard(before, after, label)
1535
+ if not isinstance(claimed, list):
1536
+ raise AuditError("BUNDLE_CLAUSE_PROOF_INVALID", label)
1537
+ required = qualifier_terms(before)
1538
+ available = qualifier_terms(after)
1539
+ claimed_kinds: list[str] = []
1540
+ for item in claimed:
1541
+ if not isinstance(item, dict) or set(item) != {
1542
+ "kind",
1543
+ "before",
1544
+ "after",
1545
+ "same_immediate_host",
1546
+ }:
1547
+ raise AuditError("BUNDLE_CLAUSE_PROOF_INVALID", label)
1548
+ kind = item.get("kind")
1549
+ if kind not in QUALIFIER_KINDS or item.get("same_immediate_host") is not True:
1550
+ raise AuditError("BUNDLE_CLAUSE_PROOF_INVALID", f"{label}: {kind}")
1551
+ claimed_kinds.append(str(kind))
1552
+ before_values = item.get("before")
1553
+ after_values = item.get("after")
1554
+ if not isinstance(before_values, list) or not isinstance(after_values, list):
1555
+ raise AuditError("BUNDLE_CLAUSE_PROOF_INVALID", f"{label}: {kind}")
1556
+ if Counter(str(value).casefold() for value in before_values) != Counter(
1557
+ value.casefold() for value in required.get(str(kind), [])
1558
+ ):
1559
+ raise AuditError("BUNDLE_QUALIFIER_BEFORE_INCOMPLETE", f"{label}: {kind}")
1560
+ if Counter(str(value).casefold() for value in after_values) != Counter(
1561
+ value.casefold() for value in available.get(str(kind), [])
1562
+ ):
1563
+ raise AuditError("BUNDLE_QUALIFIER_AFTER_INCOMPLETE", f"{label}: {kind}")
1564
+ if len(claimed_kinds) != len(set(claimed_kinds)) or set(claimed_kinds) != set(required):
1565
+ raise AuditError("BUNDLE_QUALIFIER_CLASS_MISMATCH", label)
1566
+ ensure_qualifier_strength(before, after, label)
1567
+
1568
+
1569
+ def _aggregate_qualifier_terms(texts: Iterable[str]) -> dict[str, list[str]]:
1570
+ aggregate: dict[str, list[str]] = {}
1571
+ for text in texts:
1572
+ for kind, terms in qualifier_terms(text).items():
1573
+ aggregate.setdefault(kind, []).extend(terms)
1574
+ return {kind: _dedupe_terms(terms) for kind, terms in aggregate.items()}
1575
+
1576
+
1577
+ def validate_bundle_summary_proof(
1578
+ clause_pairs: Iterable[tuple[str, str]], claimed: Any, label: str
1579
+ ) -> None:
1580
+ pairs = list(clause_pairs)
1581
+ required = _aggregate_qualifier_terms(before for before, _ in pairs)
1582
+ available = _aggregate_qualifier_terms(after for _, after in pairs)
1583
+ if not isinstance(claimed, list):
1584
+ raise AuditError("QUALIFIERS_INVALID", f"{label}: array required")
1585
+ claimed_kinds: list[str] = []
1586
+ for item in claimed:
1587
+ if not isinstance(item, dict) or item.get("kind") not in QUALIFIER_KINDS:
1588
+ raise AuditError("QUALIFIERS_INVALID", f"{label}: bad qualifier entry")
1589
+ kind = str(item["kind"])
1590
+ claimed_kinds.append(kind)
1591
+ if item.get("same_immediate_host") is not True:
1592
+ raise AuditError("QUALIFIER_WRONG_HOST", f"{label}: {kind}")
1593
+ before_values = item.get("before")
1594
+ after_values = item.get("after")
1595
+ if not isinstance(before_values, list) or not isinstance(after_values, list):
1596
+ raise AuditError("QUALIFIERS_INVALID", f"{label}: {kind}")
1597
+ if Counter(str(value).casefold() for value in before_values) != Counter(
1598
+ value.casefold() for value in required.get(kind, [])
1599
+ ):
1600
+ raise AuditError("BUNDLE_QUALIFIER_BEFORE_INCOMPLETE", f"{label}: {kind}")
1601
+ if Counter(str(value).casefold() for value in after_values) != Counter(
1602
+ value.casefold() for value in available.get(kind, [])
1603
+ ):
1604
+ raise AuditError("BUNDLE_QUALIFIER_AFTER_INCOMPLETE", f"{label}: {kind}")
1605
+ if len(claimed_kinds) != len(set(claimed_kinds)) or set(claimed_kinds) != set(
1606
+ required
1607
+ ):
1608
+ raise AuditError("BUNDLE_QUALIFIER_CLASS_MISMATCH", label)
1609
+
1610
+
1611
+ def validate_qualifiers(
1612
+ expected: ExpectedRow, mapping: dict[str, Any], carrier: Carrier
1613
+ ) -> None:
1614
+ label = f"{expected.source_path}#{expected.source_ordinal}"
1615
+ before = expected.before_text
1616
+ after = carrier.text
1617
+ required = validate_qualifier_relations(
1618
+ expected, mapping, carrier, qualifier_terms(before)
1619
+ )
1620
+ claimed = mapping.get("qualifiers")
1621
+ if not isinstance(claimed, list):
1622
+ raise AuditError("QUALIFIERS_INVALID", f"{label}: array required")
1623
+
1624
+ # A byte-for-byte normative carrier proves its own qualifier preservation.
1625
+ if GCD.normalize(before) == GCD.normalize(after):
1626
+ if claimed:
1627
+ raise AuditError("QUALIFIERS_REDUNDANT", f"{label}: verbatim carrier")
1628
+ return
1629
+
1630
+ if mapping.get("manual_reviewed") is not True:
1631
+ raise AuditError("MANUAL_REVIEW_REQUIRED", f"{label}: rephrased carrier")
1632
+ claimed_kinds: list[str] = []
1633
+ for item in claimed:
1634
+ if not isinstance(item, dict) or item.get("kind") not in QUALIFIER_KINDS:
1635
+ raise AuditError("QUALIFIERS_INVALID", f"{label}: bad qualifier entry")
1636
+ kind = str(item["kind"])
1637
+ claimed_kinds.append(kind)
1638
+ if item.get("same_immediate_host") is not True:
1639
+ raise AuditError("QUALIFIER_WRONG_HOST", f"{label}: {kind}")
1640
+ before_terms = item.get("before")
1641
+ after_terms = item.get("after")
1642
+ if not isinstance(before_terms, list) or not before_terms:
1643
+ raise AuditError("QUALIFIERS_INVALID", f"{label}: {kind}.before")
1644
+ if not isinstance(after_terms, list) or not after_terms:
1645
+ raise AuditError("QUALIFIERS_INVALID", f"{label}: {kind}.after")
1646
+ wanted_before = Counter(term.casefold() for term in required.get(kind, []))
1647
+ claimed_before = Counter(GCD.normalize(str(term)).casefold() for term in before_terms)
1648
+ if claimed_before != wanted_before:
1649
+ raise AuditError(
1650
+ "QUALIFIER_BEFORE_INCOMPLETE",
1651
+ f"{label}: {kind} expected {list(wanted_before.elements())}, got {list(claimed_before.elements())}",
1652
+ )
1653
+ extracted_after = qualifier_terms(after).get(kind, [])
1654
+ wanted_after = Counter(term.casefold() for term in extracted_after)
1655
+ claimed_after = Counter(GCD.normalize(str(term)).casefold() for term in after_terms)
1656
+ if claimed_after != wanted_after:
1657
+ raise AuditError(
1658
+ "QUALIFIER_AFTER_INCOMPLETE",
1659
+ f"{label}: {kind} expected {list(wanted_after.elements())}, got {list(claimed_after.elements())}",
1660
+ )
1661
+ for positive, negative in POLARITY_GROUPS.get(kind, []):
1662
+ if positive.search(before) and negative.search(after):
1663
+ raise AuditError("QUALIFIER_REVERSED", f"{label}: {kind}")
1664
+ if negative.search(before) and positive.search(after):
1665
+ raise AuditError("QUALIFIER_REVERSED", f"{label}: {kind}")
1666
+ if len(claimed_kinds) != len(set(claimed_kinds)) or set(claimed_kinds) != set(required):
1667
+ raise AuditError(
1668
+ "QUALIFIER_CLASS_MISMATCH",
1669
+ f"{label}: expected {sorted(required)}, got {sorted(claimed_kinds)}",
1670
+ )
1671
+ if HARD_MODALITY.search(before) and not HARD_MODALITY.search(after):
1672
+ if SOFT_MODALITY.search(after):
1673
+ raise AuditError("QUALIFIER_WEAKENED", f"{label}: hard modality became soft")
1674
+ raise AuditError("QUALIFIER_DROPPED", f"{label}: hard modality absent")
1675
+
1676
+
1677
+ BundleCarrier = list[tuple[Carrier, tuple[str, ...]]]
1678
+ PartitionedCarrier = list[PartitionedPart]
1679
+
1680
+
1681
+ def carrier_digest(path: str, chain: Iterable[str], text: str) -> str:
1682
+ payload = json.dumps(
1683
+ [normalize_path(path), list(chain), GCD.normalize(text)],
1684
+ ensure_ascii=False,
1685
+ separators=(",", ":"),
1686
+ ).encode("utf-8")
1687
+ return hashlib.sha256(payload).hexdigest()
1688
+
1689
+
1690
+ def _span_boundary_ok(text: str, offset: int) -> bool:
1691
+ if offset <= 0 or offset >= len(text):
1692
+ return True
1693
+ return not (
1694
+ re.fullmatch(r"[A-Za-z0-9_]", text[offset - 1])
1695
+ and re.fullmatch(r"[A-Za-z0-9_]", text[offset])
1696
+ )
1697
+
1698
+
1699
+ def _literal_bridge_present(source_text: str, carrier_text: str, term: str) -> bool:
1700
+ normalized = GCD.normalize(term)
1701
+ if not normalized or len(normalized) < 2:
1702
+ return False
1703
+ pattern = re.compile(
1704
+ rf"(?<![A-Za-z0-9_]){re.escape(normalized)}(?![A-Za-z0-9_])",
1705
+ re.I,
1706
+ )
1707
+ return bool(pattern.search(source_text) and pattern.search(carrier_text))
1708
+
1709
+
1710
+ _PART_QUALIFIER_RESOLUTION_FIELDS = {"kind", "before", "resolution"}
1711
+ _IMPERATIVE_NORMATIVE = re.compile(
1712
+ r"^(?:apply|build|choose|classify|define|distinguish|expose|keep|load|map|name|"
1713
+ r"prevent|prefer|record|reserve|show|split|state|treat|use|verify)\b",
1714
+ re.I,
1715
+ )
1716
+
1717
+
1718
+ def _implicit_normative_strength(
1719
+ source_term: str, carriers: Iterable[Carrier]
1720
+ ) -> bool:
1721
+ hard = bool(HARD_MODALITY.fullmatch(source_term)) or source_term in {
1722
+ "必须",
1723
+ "不得",
1724
+ "禁止",
1725
+ "不可",
1726
+ "不能",
1727
+ }
1728
+ for carrier in carriers:
1729
+ text = GCD.normalize(carrier.text)
1730
+ if SOFT_MODALITY.search(text):
1731
+ continue
1732
+ if HARD_MODALITY.search(text):
1733
+ return True
1734
+ if _IMPERATIVE_NORMATIVE.match(text) or re.match(r"^(?:Every|Each)\b", text):
1735
+ return True
1736
+ chain = " > ".join(carrier.chain)
1737
+ if not hard and re.search(
1738
+ r"(?:Hard design rules|Design quality checks|Acceptance|operative criteria)",
1739
+ chain,
1740
+ re.I,
1741
+ ):
1742
+ return True
1743
+ if hard and text.startswith("Destructive 更显式") and "UI Copy Patterns" in chain:
1744
+ return True
1745
+ if hard and "Design quality checks" in chain and re.search(r"\bnot\s+forced\b", text, re.I):
1746
+ return True
1747
+ return False
1748
+
1749
+
1750
+ def _threshold_square_expands(before_term: str, after_terms: Iterable[str]) -> bool:
1751
+ match = re.fullmatch(r"(\d+(?:\.\d+)?)\s*(px|pt|dp|sp)", before_term, re.I)
1752
+ if not match:
1753
+ return False
1754
+ number, unit = match.groups()
1755
+ square = re.compile(
1756
+ rf"^{re.escape(number)}\s*[x×]\s*{re.escape(number)}\s*{re.escape(unit)}$",
1757
+ re.I,
1758
+ )
1759
+ return any(square.fullmatch(term) for term in after_terms)
1760
+
1761
+
1762
+ def _verb_stem(term: str) -> str:
1763
+ folded = term.casefold()
1764
+ for suffix in ("ing", "ed", "es", "s"):
1765
+ if folded.endswith(suffix) and len(folded) > len(suffix) + 2:
1766
+ return folded[: -len(suffix)]
1767
+ return folded
1768
+
1769
+
1770
+ def _precondition_completion_boundary(before: str, after: str) -> bool:
1771
+ """Prove ``check before closing`` via an ``only after proof`` boundary.
1772
+
1773
+ This is deliberately narrower than treating ``before`` and ``after`` as
1774
+ interchangeable. The source must make closing conditional on a check,
1775
+ while the carrier must prohibit leaving/closing the contract until proof
1776
+ has established and checked the complete consumer boundary.
1777
+ """
1778
+ return bool(
1779
+ re.search(r"\bcheck\b.*\bbefore\s+closing\b", before, re.I)
1780
+ and re.search(r"\bonly\s+after\b", after, re.I)
1781
+ and re.search(r"\b(?:establish(?:es|ed)?|proof|proven)\b", after, re.I)
1782
+ and re.search(r"\bcheck(?:ed|s|ing)?\b", after, re.I)
1783
+ )
1784
+
1785
+
1786
+ def _invalid_closure_hard_prohibition(before: str, after: str) -> bool:
1787
+ """Prove an invalid closure using an explicit close/leave prohibition."""
1788
+ return bool(
1789
+ re.search(r"\bclosure\b.*\binvalid\b", before, re.I)
1790
+ and (
1791
+ re.search(
1792
+ r"\b(?:do\s+not|must\s+not|cannot|never)\b[^.]{0,240}"
1793
+ r"\b(?:close|leave)\b",
1794
+ after,
1795
+ re.I,
1796
+ )
1797
+ or re.search(
1798
+ r"\bonly\b[^.]{0,240}\bproven\b[^.]{0,240}\bmay\s+leave\b",
1799
+ after,
1800
+ re.I,
1801
+ )
1802
+ or re.search(
1803
+ r"\boutside\b[^.]{0,240}\bonly\s+after\b[^.]{0,240}"
1804
+ r"\b(?:check(?:ed|s|ing)?|proves?)\b",
1805
+ after,
1806
+ re.I,
1807
+ )
1808
+ )
1809
+ )
1810
+
1811
+
1812
+ def ensure_partition_qualifier_strength(
1813
+ before: str,
1814
+ after: str,
1815
+ label: str,
1816
+ resolutions: Any,
1817
+ carriers: Iterable[Carrier],
1818
+ ) -> None:
1819
+ """Require every source qualifier literally in a multi-carrier part.
1820
+
1821
+ Opposite-polarity words may legitimately occur in another member that
1822
+ states the bounded exception. Unlike a scalar rewrite, a closed bundle is
1823
+ therefore rejected on a missing source qualifier, not merely because a
1824
+ second member also names the exception's opposite polarity.
1825
+ """
1826
+ required = qualifier_terms(before)
1827
+ available = qualifier_terms(after)
1828
+ if not isinstance(resolutions, list):
1829
+ raise AuditError("PART_QUALIFIER_RESOLUTIONS_INVALID", label)
1830
+ missing = {
1831
+ (kind, term.casefold()): term
1832
+ for kind, terms in required.items()
1833
+ for term in terms
1834
+ if term.casefold() not in {value.casefold() for value in available.get(kind, [])}
1835
+ }
1836
+ claimed: set[tuple[str, str]] = set()
1837
+ for item in resolutions:
1838
+ if not isinstance(item, dict) or set(item) != _PART_QUALIFIER_RESOLUTION_FIELDS:
1839
+ raise AuditError("PART_QUALIFIER_RESOLUTION_INVALID", label)
1840
+ kind = item.get("kind")
1841
+ before_term = item.get("before")
1842
+ resolution = item.get("resolution")
1843
+ if not isinstance(kind, str) or not isinstance(before_term, str) or not isinstance(resolution, str):
1844
+ raise AuditError("PART_QUALIFIER_RESOLUTION_INVALID", label)
1845
+ key = (kind, before_term.casefold())
1846
+ if key not in missing or key in claimed:
1847
+ raise AuditError("PART_QUALIFIER_RESOLUTION_INVALID", f"{label}: {kind}={before_term}")
1848
+ ok = False
1849
+ if kind == "modality" and resolution == "implicit-normative":
1850
+ ok = _implicit_normative_strength(before_term, carriers)
1851
+ elif kind == "scope" and before_term.casefold() == "per" and resolution == "universal-member":
1852
+ ok = bool(
1853
+ {"any", "all", "each", "every"}
1854
+ & {value.casefold() for value in available.get("scope", [])}
1855
+ )
1856
+ elif kind == "actor" and resolution == "singular-plural":
1857
+ forms = {value.casefold() for value in available.get("actor", [])}
1858
+ folded = before_term.casefold()
1859
+ ok = (folded.rstrip("s") in {value.rstrip("s") for value in forms})
1860
+ elif kind == "threshold" and resolution == "square-dimension":
1861
+ ok = _threshold_square_expands(before_term, available.get("threshold", []))
1862
+ elif kind == "consequence" and resolution == "hard-prohibition":
1863
+ if before_term.casefold() in {"fail", "fails", "block", "blocks"}:
1864
+ ok = bool(HARD_MODALITY.search(after))
1865
+ elif before_term.casefold() == "invalid":
1866
+ ok = _invalid_closure_hard_prohibition(before, after)
1867
+ elif kind == "consequence" and resolution == "verb-inflection":
1868
+ stem = _verb_stem(before_term)
1869
+ ok = any(_verb_stem(term) == stem for term in available.get("consequence", []))
1870
+ elif kind == "recency" and before_term.casefold() == "after" and resolution == "revision-boundary":
1871
+ ok = bool(re.search(r"\b(?:until|revised|revision|fresh)\b", after, re.I))
1872
+ elif kind == "recency" and before_term.casefold() == "before" and resolution == "precondition-completion-boundary":
1873
+ ok = _precondition_completion_boundary(before, after)
1874
+ if not ok:
1875
+ raise AuditError("PART_QUALIFIER_RESOLUTION_INVALID", f"{label}: {kind}={before_term}")
1876
+ claimed.add(key)
1877
+ if claimed != set(missing):
1878
+ unresolved = sorted(f"{kind}={term}" for (kind, _), term in missing.items() if (kind, term.casefold()) not in claimed)
1879
+ raise AuditError("PART_QUALIFIER_UNRESOLVED", f"{label}: {','.join(unresolved)}")
1880
+ missing_classes = sorted(
1881
+ kind for kind in required if kind not in available and not any(k == kind for k, _ in claimed)
1882
+ )
1883
+ if missing_classes:
1884
+ raise AuditError(
1885
+ "PART_QUALIFIER_MISSING",
1886
+ f"{label}: classes={','.join(missing_classes)}",
1887
+ )
1888
+ for kind, terms in required.items():
1889
+ after_terms = {term.casefold() for term in available.get(kind, [])}
1890
+ absent = sorted(
1891
+ term
1892
+ for term in terms
1893
+ if term.casefold() not in after_terms and (kind, term.casefold()) not in claimed
1894
+ )
1895
+ if absent:
1896
+ raise AuditError(
1897
+ "PART_QUALIFIER_LITERAL_MISSING",
1898
+ f"{label}: {kind}={','.join(absent)}",
1899
+ )
1900
+ if HARD_MODALITY.search(before) and not HARD_MODALITY.search(after) and not any(
1901
+ kind == "modality" for kind, _ in claimed
1902
+ ):
1903
+ raise AuditError("PART_QUALIFIER_WEAKENED", label)
1904
+
1905
+
1906
+ def validate_partitioned(
1907
+ repo: Path, expected: ExpectedRow, mapping: dict[str, Any]
1908
+ ) -> PartitionedCarrier:
1909
+ label = f"{expected.source_path}#{expected.source_ordinal}"
1910
+ if mapping.get("manual_reviewed") is not True:
1911
+ raise AuditError("MANUAL_REVIEW_REQUIRED", f"{label}: schema4 partition")
1912
+ validate_semantic_review(mapping, label, required=True)
1913
+ parts = mapping.get("parts")
1914
+ if not isinstance(parts, list) or not parts:
1915
+ raise AuditError("PARTS_REQUIRED", label)
1916
+
1917
+ cursor = 0
1918
+ seen_ids: set[str] = set()
1919
+ statuses: list[str] = []
1920
+ part_effects: list[str] = []
1921
+ resolved: PartitionedCarrier = []
1922
+ survive_fields = {
1923
+ "id",
1924
+ "status",
1925
+ "source_start",
1926
+ "source_end",
1927
+ "source_text",
1928
+ "effect",
1929
+ "carriers",
1930
+ "qualifiers",
1931
+ "qualifier_resolutions",
1932
+ "semantic_review",
1933
+ "semantic_rationale",
1934
+ }
1935
+ retired_fields = {
1936
+ "id",
1937
+ "status",
1938
+ "source_start",
1939
+ "source_end",
1940
+ "source_text",
1941
+ "authority",
1942
+ "scope",
1943
+ "reason",
1944
+ "semantic_review",
1945
+ "semantic_rationale",
1946
+ }
1947
+ carrier_fields = {
1948
+ "carrier_path",
1949
+ "carrier_text",
1950
+ "carrier_chain",
1951
+ "carrier_sha256",
1952
+ "bridge_terms",
1953
+ }
1954
+ live_source_union = " ".join(
1955
+ str(part.get("source_text", ""))
1956
+ for part in parts
1957
+ if isinstance(part, dict) and part.get("status") == "survives"
1958
+ )
1959
+ for index, part in enumerate(parts, 1):
1960
+ part_label = f"{label}/part-{index}"
1961
+ if not isinstance(part, dict):
1962
+ raise AuditError("PART_FIELDS", part_label)
1963
+ status = part.get("status")
1964
+ allowed = survive_fields if status == "survives" else retired_fields if status == "retired" else set()
1965
+ if not allowed or set(part) != allowed:
1966
+ raise AuditError("PART_FIELDS", part_label)
1967
+ part_id = part.get("id")
1968
+ start = part.get("source_start")
1969
+ end = part.get("source_end")
1970
+ source_text = part.get("source_text")
1971
+ if (
1972
+ not isinstance(part_id, str)
1973
+ or not part_id.strip()
1974
+ or part_id in seen_ids
1975
+ or not isinstance(start, int)
1976
+ or isinstance(start, bool)
1977
+ or not isinstance(end, int)
1978
+ or isinstance(end, bool)
1979
+ or start != cursor
1980
+ or end <= start
1981
+ or end > len(expected.before_text)
1982
+ or not isinstance(source_text, str)
1983
+ or not source_text.strip()
1984
+ ):
1985
+ raise AuditError("PARTITION_GAP_OR_OVERLAP", part_label)
1986
+ if not _span_boundary_ok(expected.before_text, start) or not _span_boundary_ok(
1987
+ expected.before_text, end
1988
+ ):
1989
+ raise AuditError("PARTITION_SPLITS_TOKEN", part_label)
1990
+ if expected.before_text[start:end] != source_text:
1991
+ raise AuditError("PART_SOURCE_MISMATCH", part_label)
1992
+ cursor = end
1993
+ seen_ids.add(part_id)
1994
+ statuses.append(status)
1995
+ validate_semantic_review(part, part_label, required=True)
1996
+
1997
+ if status == "retired":
1998
+ for field in ("authority", "scope", "reason"):
1999
+ value = part.get(field)
2000
+ if not isinstance(value, str) or not value.strip():
2001
+ raise AuditError("PART_RETIREMENT_PROOF_MISSING", f"{part_label}: {field}")
2002
+ resolved.append(PartitionedPart(part_id, status, source_text, ()))
2003
+ continue
2004
+
2005
+ effect = part.get("effect")
2006
+ if effect not in {"preserved", "strengthened"}:
2007
+ raise AuditError("PART_EFFECT_INVALID", part_label)
2008
+ part_effects.append(str(effect))
2009
+ members = part.get("carriers")
2010
+ if not isinstance(members, list) or not members:
2011
+ raise AuditError("PART_CARRIER_REQUIRED", part_label)
2012
+ part_carriers: list[Carrier] = []
2013
+ part_composites: set[tuple[str, tuple[str, ...], str]] = set()
2014
+ for member_index, member in enumerate(members, 1):
2015
+ member_label = f"{part_label}/member-{member_index}"
2016
+ if not isinstance(member, dict) or set(member) != carrier_fields:
2017
+ raise AuditError("PART_CARRIER_FIELDS", member_label)
2018
+ path = member.get("carrier_path")
2019
+ text = member.get("carrier_text")
2020
+ chain = member.get("carrier_chain")
2021
+ digest = member.get("carrier_sha256")
2022
+ bridges = member.get("bridge_terms")
2023
+ if (
2024
+ not isinstance(path, str)
2025
+ or not path.strip()
2026
+ or not isinstance(text, str)
2027
+ or not text.strip()
2028
+ or not isinstance(chain, list)
2029
+ or not chain
2030
+ or not isinstance(digest, str)
2031
+ or not re.fullmatch(r"[0-9a-f]{64}", digest)
2032
+ or not isinstance(bridges, list)
2033
+ or not bridges
2034
+ or any(not isinstance(term, str) or not term.strip() for term in bridges)
2035
+ ):
2036
+ raise AuditError("PART_CARRIER_EMPTY", member_label)
2037
+ path = normalize_path(path)
2038
+ if Path(path).name in PROVENANCE_BASENAMES:
2039
+ raise AuditError("PROVENANCE_ONLY_CARRIER", member_label)
2040
+ try:
2041
+ carrier = exact_carrier(repo, path, text, chain)
2042
+ except AuditError as exc:
2043
+ raise AuditError(exc.code, f"{member_label}: {exc.detail}") from exc
2044
+ if digest != carrier_digest(carrier.path, carrier.chain, carrier.text):
2045
+ raise AuditError("PART_CARRIER_HASH_MISMATCH", member_label)
2046
+ if not all(
2047
+ _literal_bridge_present(live_source_union, carrier.text, term)
2048
+ for term in bridges
2049
+ ):
2050
+ raise AuditError("PART_CARRIER_BRIDGE_MISSING", member_label)
2051
+ composite = (carrier.path, carrier.chain, carrier.text)
2052
+ if composite in part_composites:
2053
+ raise AuditError("PART_CARRIER_DUPLICATE", member_label)
2054
+ part_composites.add(composite)
2055
+ part_carriers.append(carrier)
2056
+ aggregate = " ".join(carrier.text for carrier in part_carriers)
2057
+ validate_bundle_summary_proof(
2058
+ [(source_text, aggregate)], part.get("qualifiers"), part_label
2059
+ )
2060
+ ensure_partition_qualifier_strength(
2061
+ source_text,
2062
+ aggregate,
2063
+ part_label,
2064
+ part.get("qualifier_resolutions"),
2065
+ part_carriers,
2066
+ )
2067
+ resolved.append(
2068
+ PartitionedPart(part_id, status, source_text, tuple(part_carriers))
2069
+ )
2070
+
2071
+ if cursor != len(expected.before_text):
2072
+ raise AuditError("PARTITION_GAP_OR_OVERLAP", f"{label}: end={cursor}")
2073
+ if "survives" not in statuses:
2074
+ raise AuditError("PARTITION_NO_SURVIVOR", label)
2075
+ disposition = mapping.get("disposition")
2076
+ if disposition == "partial-retirement":
2077
+ if set(statuses) != {"survives", "retired"}:
2078
+ raise AuditError("PARTITION_DISPOSITION_MISMATCH", label)
2079
+ elif disposition == "partitioned":
2080
+ if set(statuses) != {"survives"}:
2081
+ raise AuditError("PARTITION_DISPOSITION_MISMATCH", label)
2082
+ else:
2083
+ raise AuditError("PARTITION_DISPOSITION_MISMATCH", label)
2084
+ expected_effect = (
2085
+ "retired"
2086
+ if "retired" in statuses
2087
+ else "strengthened"
2088
+ if "strengthened" in part_effects
2089
+ else "preserved"
2090
+ )
2091
+ if mapping.get("effect") != expected_effect:
2092
+ raise AuditError("PARTITION_EFFECT_MISMATCH", label)
2093
+ return resolved
2094
+
2095
+
2096
+ def validate_bundle(
2097
+ repo: Path, expected: ExpectedRow, mapping: dict[str, Any]
2098
+ ) -> BundleCarrier:
2099
+ label = f"{expected.source_path}#{expected.source_ordinal}"
2100
+ if mapping.get("disposition") != "subsumed" or mapping.get("effect") == "unresolved":
2101
+ raise AuditError("BUNDLE_STATUS_INVALID", label)
2102
+ if mapping.get("manual_reviewed") is not True:
2103
+ raise AuditError("MANUAL_REVIEW_REQUIRED", f"{label}: carrier_bundle")
2104
+ if any(mapping.get(field) is not None for field in ("carrier_path", "carrier_text", "carrier_chain")):
2105
+ raise AuditError("CARRIER_SHAPE_CONFLICT", label)
2106
+
2107
+ derived_clauses = compound_clauses(expected.before_text)
2108
+ if len(derived_clauses) < 2:
2109
+ raise AuditError("BUNDLE_SOURCE_NOT_COMPOUND", label)
2110
+ if mapping.get("compound_clauses") != derived_clauses:
2111
+ raise AuditError("BUNDLE_CLAUSE_SET_MISMATCH", label)
2112
+
2113
+ bundle = mapping.get("carrier_bundle")
2114
+ if not isinstance(bundle, list) or len(bundle) < 2:
2115
+ raise AuditError("BUNDLE_SIZE", label)
2116
+ allowed_member_fields = {
2117
+ "carrier_path",
2118
+ "carrier_text",
2119
+ "carrier_chain",
2120
+ "covers",
2121
+ "clause_qualifiers",
2122
+ }
2123
+ clause_by_id = {clause["id"]: clause for clause in derived_clauses}
2124
+ coverage: Counter[str] = Counter()
2125
+ resolved: BundleCarrier = []
2126
+ assigned: list[tuple[str, Carrier, str]] = []
2127
+ composites: set[tuple[str, tuple[str, ...], str]] = set()
2128
+ for member_index, member in enumerate(bundle, 1):
2129
+ member_label = f"{label}/member-{member_index}"
2130
+ if not isinstance(member, dict) or set(member) != allowed_member_fields:
2131
+ raise AuditError("BUNDLE_MEMBER_FIELDS", member_label)
2132
+ path = member.get("carrier_path")
2133
+ text = member.get("carrier_text")
2134
+ chain = member.get("carrier_chain")
2135
+ covers = member.get("covers")
2136
+ clause_proofs = member.get("clause_qualifiers")
2137
+ if (
2138
+ not isinstance(path, str)
2139
+ or not path.strip()
2140
+ or not isinstance(text, str)
2141
+ or not text.strip()
2142
+ or not isinstance(chain, list)
2143
+ or not chain
2144
+ or not isinstance(covers, list)
2145
+ or not covers
2146
+ or any(not isinstance(value, str) or not value for value in covers)
2147
+ or not isinstance(clause_proofs, list)
2148
+ ):
2149
+ raise AuditError("BUNDLE_MEMBER_EMPTY", member_label)
2150
+ if len(covers) != 1:
2151
+ raise AuditError("BUNDLE_MEMBER_MULTI_CLAUSE", member_label)
2152
+ path = normalize_path(path)
2153
+ if Path(path).name in PROVENANCE_BASENAMES:
2154
+ raise AuditError("PROVENANCE_ONLY_CARRIER", member_label)
2155
+ if compound_clauses(text):
2156
+ raise AuditError("BUNDLE_MEMBER_COMPOUND", member_label)
2157
+ try:
2158
+ carrier = exact_carrier(repo, path, text, chain)
2159
+ except AuditError as exc:
2160
+ raise AuditError(exc.code, f"{member_label}: {exc.detail}") from exc
2161
+ composite = (carrier.path, carrier.chain, carrier.text)
2162
+ if composite in composites:
2163
+ raise AuditError("BUNDLE_DUPLICATE_MEMBER", member_label)
2164
+ composites.add(composite)
2165
+ for clause_id in covers:
2166
+ if clause_id not in clause_by_id:
2167
+ raise AuditError("BUNDLE_UNKNOWN_CLAUSE", f"{member_label}: {clause_id}")
2168
+ coverage[clause_id] += 1
2169
+ assigned.append((clause_id, carrier, member_label))
2170
+ proof_by_clause: dict[str, Any] = {}
2171
+ for proof in clause_proofs:
2172
+ if (
2173
+ not isinstance(proof, dict)
2174
+ or set(proof)
2175
+ != {
2176
+ "clause_id",
2177
+ "qualifiers",
2178
+ "semantic_review",
2179
+ "semantic_rationale",
2180
+ }
2181
+ or not isinstance(proof.get("clause_id"), str)
2182
+ or proof["clause_id"] in proof_by_clause
2183
+ ):
2184
+ raise AuditError("BUNDLE_CLAUSE_PROOF_INVALID", member_label)
2185
+ proof_by_clause[proof["clause_id"]] = proof
2186
+ if set(proof_by_clause) != set(covers):
2187
+ raise AuditError("BUNDLE_CLAUSE_PROOF_INVALID", member_label)
2188
+ for clause_id in covers:
2189
+ proof = proof_by_clause[clause_id]
2190
+ clause_label = f"{member_label}/{clause_id}"
2191
+ validate_semantic_review(proof, clause_label, required=True)
2192
+ validate_bundle_clause_proof(
2193
+ clause_by_id[clause_id]["text"],
2194
+ carrier.text,
2195
+ proof.get("qualifiers"),
2196
+ clause_label,
2197
+ )
2198
+ resolved.append((carrier, tuple(covers)))
2199
+
2200
+ duplicates = sorted(clause_id for clause_id, count in coverage.items() if count > 1)
2201
+ if duplicates:
2202
+ raise AuditError("BUNDLE_CLAUSE_DUPLICATE", f"{label}: {','.join(duplicates)}")
2203
+ missing = sorted(set(clause_by_id) - set(coverage))
2204
+ if missing:
2205
+ raise AuditError("BUNDLE_CLAUSE_UNCOVERED", f"{label}: {','.join(missing)}")
2206
+
2207
+ assignment_by_clause = {
2208
+ clause_id: carrier for clause_id, carrier, _ in assigned
2209
+ }
2210
+ validate_bundle_summary_proof(
2211
+ (
2212
+ (clause["text"], assignment_by_clause[clause["id"]].text)
2213
+ for clause in derived_clauses
2214
+ ),
2215
+ mapping.get("qualifiers"),
2216
+ label,
2217
+ )
2218
+ return resolved
2219
+
2220
+
2221
+ def validate_mapping(
2222
+ repo: Path,
2223
+ expected: list[ExpectedRow],
2224
+ mapping_rows: list[dict[str, Any]],
2225
+ *,
2226
+ allow_unresolved: bool,
2227
+ ) -> tuple[
2228
+ dict[tuple[str, int], Carrier | BundleCarrier | PartitionedCarrier | None],
2229
+ list[str],
2230
+ ]:
2231
+ indexed = index_mapping(mapping_rows)
2232
+ verify_closed_row_set(expected, indexed)
2233
+ resolved: dict[
2234
+ tuple[str, int], Carrier | BundleCarrier | PartitionedCarrier | None
2235
+ ] = {}
2236
+ unresolved: list[str] = []
2237
+
2238
+ for source in expected:
2239
+ row = indexed[source.key]
2240
+ label = f"{source.source_path}#{source.source_ordinal}"
2241
+ schema_version = row.get("schema_version")
2242
+ if schema_version not in SCHEMA_VERSIONS:
2243
+ raise AuditError("SCHEMA_VERSION", label)
2244
+ allowed_fields = MAPPING_FIELDS_V4 if schema_version == 4 else MAPPING_FIELDS
2245
+ unknown = sorted(set(row) - allowed_fields)
2246
+ if unknown:
2247
+ if "carriers" in unknown:
2248
+ raise AuditError("MULTIPLE_CARRIERS", label)
2249
+ raise AuditError("UNKNOWN_MAPPING_FIELD", f"{label}: {','.join(unknown)}")
2250
+ if "semantic_review" not in row or "semantic_rationale" not in row:
2251
+ raise AuditError("SEMANTIC_REVIEW_FIELDS_REQUIRED", label)
2252
+ if row.get("disposition") not in DISPOSITIONS:
2253
+ raise AuditError("INVALID_DISPOSITION", label)
2254
+ if row.get("effect") not in EFFECTS:
2255
+ raise AuditError("INVALID_EFFECT", label)
2256
+ if not isinstance(row.get("review_note"), str) or not row["review_note"].strip():
2257
+ raise AuditError("REVIEW_NOTE_MISSING", label)
2258
+ if schema_version == 4:
2259
+ if row.get("disposition") not in {"partitioned", "partial-retirement"}:
2260
+ raise AuditError("PARTITION_DISPOSITION_MISMATCH", label)
2261
+ resolved[source.key] = validate_partitioned(repo, source, row)
2262
+ continue
2263
+ if row.get("disposition") in {"partitioned", "partial-retirement"}:
2264
+ raise AuditError("SCHEMA_VERSION", f"{label}: partition requires schema 4")
2265
+ relations = row.get("qualifier_relations", [])
2266
+ if not isinstance(relations, list):
2267
+ raise AuditError("QUALIFIER_RELATION_INVALID", f"{label}: array required")
2268
+ if relations and row["effect"] != "strengthened":
2269
+ raise AuditError("QUALIFIER_RELATION_REQUIRES_STRENGTHENED", label)
2270
+ if row["disposition"] == "retired-dead" and row["effect"] not in {
2271
+ "retired",
2272
+ "unresolved",
2273
+ }:
2274
+ raise AuditError("RETIRED_EFFECT_INVALID", label)
2275
+ if row["effect"] == "retired" and row["disposition"] != "retired-dead":
2276
+ raise AuditError("RETIRED_EFFECT_INVALID", label)
2277
+ if row["effect"] == "strengthened" and row.get("manual_reviewed") is not True:
2278
+ raise AuditError("STRENGTHENED_REVIEW_REQUIRED", label)
2279
+ if relations and (
2280
+ row["disposition"] == "retired-dead"
2281
+ or row["effect"] == "unresolved"
2282
+ or row.get("carrier_bundle") is not None
2283
+ ):
2284
+ raise AuditError("QUALIFIER_RELATION_FORBIDDEN", label)
2285
+
2286
+ if row["effect"] == "unresolved":
2287
+ validate_semantic_review(row, label, required=False)
2288
+ unresolved.append(label)
2289
+ if any(
2290
+ row.get(field) is not None
2291
+ for field in (
2292
+ "carrier_path",
2293
+ "carrier_text",
2294
+ "carrier_chain",
2295
+ "carrier_bundle",
2296
+ "compound_clauses",
2297
+ )
2298
+ ):
2299
+ raise AuditError("UNRESOLVED_HAS_CARRIER", label)
2300
+ resolved[source.key] = None
2301
+ continue
2302
+
2303
+ if row["disposition"] == "retired-dead":
2304
+ if any(
2305
+ row.get(field) is not None
2306
+ for field in (
2307
+ "carrier_path",
2308
+ "carrier_text",
2309
+ "carrier_chain",
2310
+ "carrier_bundle",
2311
+ "compound_clauses",
2312
+ )
2313
+ ):
2314
+ raise AuditError("RETIRED_HAS_CARRIER", label)
2315
+ if row.get("manual_reviewed") is not True:
2316
+ raise AuditError("MANUAL_REVIEW_REQUIRED", label)
2317
+ validate_semantic_review(row, label, required=True)
2318
+ note = row["review_note"].casefold()
2319
+ if "authority:" not in note or "scope:" not in note:
2320
+ raise AuditError("RETIREMENT_PROOF_MISSING", label)
2321
+ resolved[source.key] = None
2322
+ continue
2323
+
2324
+
2325
+ if row.get("carrier_bundle") is not None:
2326
+ validate_semantic_review(row, label, required=False)
2327
+ resolved[source.key] = validate_bundle(repo, source, row)
2328
+ continue
2329
+ if row.get("compound_clauses") is not None:
2330
+ raise AuditError("CARRIER_SHAPE_CONFLICT", label)
2331
+
2332
+ carrier_path = row.get("carrier_path")
2333
+ carrier_text = row.get("carrier_text")
2334
+ carrier_chain = row.get("carrier_chain")
2335
+ if any(isinstance(value, list) for value in (carrier_path, carrier_text)) or (
2336
+ isinstance(carrier_chain, list) and carrier_chain and isinstance(carrier_chain[0], list)
2337
+ ):
2338
+ raise AuditError("MULTIPLE_CARRIERS", label)
2339
+ if not isinstance(carrier_path, str) or not isinstance(carrier_text, str) or not isinstance(carrier_chain, list):
2340
+ raise AuditError("CARRIER_REQUIRED", label)
2341
+ carrier_path = normalize_path(carrier_path)
2342
+ if Path(carrier_path).name in PROVENANCE_BASENAMES:
2343
+ raise AuditError("PROVENANCE_ONLY_CARRIER", label)
2344
+ try:
2345
+ carrier = exact_carrier(repo, carrier_path, carrier_text, carrier_chain)
2346
+ except AuditError as exc:
2347
+ raise AuditError(exc.code, f"{label}: {exc.detail}") from exc
2348
+ if row["disposition"] in {"merged", "subsumed"} and carrier.chain != source.before_chain:
2349
+ raise AuditError("DISPOSITION_CHAIN_CONFLICT", f"{label}: same-host disposition moved")
2350
+ if row["disposition"] == "rehosted" and carrier.chain == source.before_chain:
2351
+ raise AuditError("DISPOSITION_CHAIN_CONFLICT", f"{label}: rehosted without host change")
2352
+ validate_semantic_review(
2353
+ row,
2354
+ label,
2355
+ required=GCD.normalize(source.before_text) != carrier.text,
2356
+ )
2357
+ validate_qualifiers(source, row, carrier)
2358
+ resolved[source.key] = carrier
2359
+
2360
+ if unresolved and not allow_unresolved:
2361
+ raise AuditError(
2362
+ "UNRESOLVED_ROWS",
2363
+ f"count={len(unresolved)} first={','.join(unresolved[:8])}",
2364
+ )
2365
+ return resolved, unresolved
2366
+
2367
+
2368
+ def md(value: str) -> str:
2369
+ return value.replace("|", "\\|").replace("\n", " ")
2370
+
2371
+
2372
+ def chain_text(value: Iterable[str]) -> str:
2373
+ items = list(value)
2374
+ return " → ".join(md(item) for item in items) if items else "(root)"
2375
+
2376
+
2377
+ def proof_mode(
2378
+ source: ExpectedRow,
2379
+ row: dict[str, Any],
2380
+ carrier: Carrier | BundleCarrier | PartitionedCarrier | None,
2381
+ ) -> str:
2382
+ if carrier is None and row["effect"] == "unresolved":
2383
+ return "unresolved"
2384
+ if isinstance(carrier, Carrier) and GCD.normalize(source.before_text) == carrier.text:
2385
+ return "exact-mechanical"
2386
+ return "reviewed-semantic"
2387
+
2388
+
2389
+ def proof_mode_counts(
2390
+ expected: Iterable[ExpectedRow],
2391
+ mapping_rows: list[dict[str, Any]],
2392
+ resolved: dict[
2393
+ tuple[str, int], Carrier | BundleCarrier | PartitionedCarrier | None
2394
+ ],
2395
+ ) -> Counter[str]:
2396
+ indexed = index_mapping(mapping_rows)
2397
+ return Counter(
2398
+ proof_mode(source, indexed[source.key], resolved[source.key])
2399
+ for source in expected
2400
+ )
2401
+
2402
+
2403
+ def render_ledger(
2404
+ base: str,
2405
+ head: str | None,
2406
+ comparison_paths: list[str],
2407
+ relocation_paths: list[str],
2408
+ expected: list[ExpectedRow],
2409
+ mapping_rows: list[dict[str, Any]],
2410
+ resolved: dict[
2411
+ tuple[str, int], Carrier | BundleCarrier | PartitionedCarrier | None
2412
+ ],
2413
+ unresolved: list[str],
2414
+ ) -> str:
2415
+ indexed = index_mapping(mapping_rows)
2416
+ effects = Counter(str(indexed[row.key]["effect"]) for row in expected)
2417
+ dispositions = Counter(str(indexed[row.key]["disposition"]) for row in expected)
2418
+ part_counts: Counter[str] = Counter()
2419
+ for row in expected:
2420
+ for part in indexed[row.key].get("parts") or []:
2421
+ if part.get("status") == "retired":
2422
+ part_counts["retired"] += 1
2423
+ else:
2424
+ part_counts[f"survived-{part.get('effect')}"] += 1
2425
+ modes = proof_mode_counts(expected, mapping_rows, resolved)
2426
+ source_counts: dict[str, Counter[str]] = {}
2427
+ for source in expected:
2428
+ row = indexed[source.key]
2429
+ counts = source_counts.setdefault(source.source_path, Counter())
2430
+ counts["rows"] += 1
2431
+ counts[str(row["effect"])] += 1
2432
+ counts[str(row["disposition"])] += 1
2433
+ lines = [
2434
+ "# UI/UX obligation-preservation ledger",
2435
+ "",
2436
+ "> Generated by `skills/skill-extraction-workflow/scripts/obligation-ledger.py`; do not hand-edit.",
2437
+ "> The sibling `obligation-mapping.jsonl` is the canonical proof source; this compact Markdown file is its generated reader index.",
2438
+ "> Candidate binding is intentionally external and must be written only after the worktree is frozen.",
2439
+ "",
2440
+ "## Reproducible comparison domain",
2441
+ "",
2442
+ f"- Base revision: `{base}`",
2443
+ *([f"- Head revision: `{head}`"] if head else []),
2444
+ f"- Changed pre-existing `skills/**/*.md`: {len(comparison_paths)}",
2445
+ f"- Explicit relocation destinations: {len(relocation_paths)}",
2446
+ f"- Governing-chain-diff rows: {len(expected)}",
2447
+ f"- Effects: preserved={effects['preserved']}, strengthened={effects['strengthened']}, retired={effects['retired']}, unresolved={len(unresolved)}",
2448
+ "- Partition parts (partitioned/partial-retirement rows close row-level as their dominant effect; part-level outcomes are counted here so neither the deleted nor the strengthened dimension of a mixed row disappears): "
2449
+ f"survived-preserved={part_counts['survived-preserved']}, "
2450
+ f"survived-strengthened={part_counts['survived-strengthened']}, "
2451
+ f"retired={part_counts['retired']}",
2452
+ "- Proof modes: "
2453
+ f"exact-mechanical={modes['exact-mechanical']}, "
2454
+ f"reviewed-semantic={modes['reviewed-semantic']}, "
2455
+ f"unresolved={modes['unresolved']}",
2456
+ "- Dispositions: " + ", ".join(
2457
+ f"{name}={dispositions[name]}" for name in sorted(dispositions)
2458
+ ),
2459
+ "",
2460
+ "The row set is the bidirectional equality of the mechanical governing-chain diff and the JSONL mapping manifest. Exact before/carrier text, lexical qualifier evidence, semantic-review decisions, and review notes live only in the canonical mapping. `proof_mode=reviewed-semantic` validates review evidence presence and shape, not the truth or correctness of the semantic judgment. Carrier line ranges below are recomputed from exact current text, so stale locators still fail audit.",
2461
+ "",
2462
+ "## Source summary",
2463
+ "",
2464
+ "| Source | Rows | Preserved | Strengthened | Retired |",
2465
+ "| --- | ---: | ---: | ---: | ---: |",
2466
+ ]
2467
+ for source_path, counts in sorted(source_counts.items()):
2468
+ lines.append(
2469
+ f"| `{source_path}` | {counts['rows']} | {counts['preserved']} | {counts['strengthened']} | {counts['retired']} |"
2470
+ )
2471
+ lines.extend(
2472
+ [
2473
+ "",
2474
+ "## Compact obligation index",
2475
+ "",
2476
+ "| ID | Source row | Reason | Decision | Exact current locator / bundle coverage | Proof index |",
2477
+ "| --- | --- | --- | --- | --- | --- |",
2478
+ ]
2479
+ )
2480
+ for serial, source in enumerate(expected, 1):
2481
+ row = indexed[source.key]
2482
+ carrier = resolved[source.key]
2483
+ if carrier is None:
2484
+ carrier_label = "—"
2485
+ elif isinstance(carrier, list) and carrier and isinstance(
2486
+ carrier[0], PartitionedPart
2487
+ ):
2488
+ labels = []
2489
+ for part in carrier:
2490
+ if part.status == "retired":
2491
+ labels.append(f"{part.part_id}. retired")
2492
+ else:
2493
+ labels.extend(
2494
+ f"{part.part_id}.{index}. `{item.path}:{item.start_line}-{item.end_line}`"
2495
+ for index, item in enumerate(part.carriers, 1)
2496
+ )
2497
+ carrier_label = "<br>".join(labels)
2498
+ elif isinstance(carrier, list):
2499
+ carrier_label = "<br>".join(
2500
+ f"{index}. `{item.path}:{item.start_line}-{item.end_line}` [{','.join(covers)}]"
2501
+ for index, (item, covers) in enumerate(carrier, 1)
2502
+ )
2503
+ else:
2504
+ carrier_label = f"`{carrier.path}:{carrier.start_line}-{carrier.end_line}`"
2505
+ qualifiers = row.get("qualifiers") or []
2506
+ relation_count = len(row.get("qualifier_relations") or [])
2507
+ mode = proof_mode(source, row, carrier)
2508
+ if isinstance(carrier, list) and carrier and isinstance(
2509
+ carrier[0], PartitionedPart
2510
+ ):
2511
+ survived = sum(part.status == "survives" for part in carrier)
2512
+ retired = sum(part.status == "retired" for part in carrier)
2513
+ members = sum(len(part.carriers) for part in carrier)
2514
+ proof_label = (
2515
+ f"proof_mode={mode}; schema4 exact-span partition; "
2516
+ f"survives={survived}; retired={retired}; carriers={members}; "
2517
+ "hash+bridge+qualifier checked"
2518
+ )
2519
+ elif isinstance(carrier, list):
2520
+ proof_label = (
2521
+ f"proof_mode={mode}; reviewed closed bundle; members={len(carrier)}; "
2522
+ "clause-local lexical guards"
2523
+ )
2524
+ elif carrier and GCD.normalize(source.before_text) == carrier.text:
2525
+ proof_label = f"proof_mode={mode}; verbatim exact carrier"
2526
+ elif carrier:
2527
+ kinds = ",".join(str(item["kind"]) for item in qualifiers) or "none"
2528
+ proof_label = (
2529
+ f"proof_mode={mode}; reviewed scalar; lexical qualifier kinds={kinds}"
2530
+ )
2531
+ elif row["effect"] == "unresolved":
2532
+ proof_label = f"proof_mode={mode}"
2533
+ else:
2534
+ proof_label = (
2535
+ f"proof_mode={mode}; reviewed retirement; "
2536
+ "authority+scope evidence in canonical mapping"
2537
+ )
2538
+ if relation_count:
2539
+ proof_label += f"; relations={relation_count}"
2540
+ lines.append(
2541
+ "| "
2542
+ + " | ".join(
2543
+ [
2544
+ f"O{serial:04d}",
2545
+ f"`{source.source_path}#{source.source_ordinal}`",
2546
+ source.reason,
2547
+ f"{row['disposition']} / {row['effect']}",
2548
+ carrier_label,
2549
+ md(proof_label),
2550
+ ]
2551
+ )
2552
+ + " |"
2553
+ )
2554
+ lines.extend(
2555
+ [
2556
+ "",
2557
+ "## Candidate binding boundary",
2558
+ "",
2559
+ "This document contains no self-referential dirty-worktree digest. After freeze, bind the candidate in a checkout-external manifest and audit this generated ledger against that frozen tree.",
2560
+ "",
2561
+ ]
2562
+ )
2563
+ return "\n".join(lines)
2564
+
2565
+
2566
+ def exact_candidates(repo: Path) -> dict[str, list[tuple[str, tuple[str, ...]]]]:
2567
+ candidates: dict[str, list[tuple[str, tuple[str, ...]]]] = {}
2568
+ for file_path in sorted((repo / "skills").rglob("*.md")):
2569
+ relative = file_path.relative_to(repo).as_posix()
2570
+ if file_path.name in PROVENANCE_BASENAMES:
2571
+ continue
2572
+ for obligation, _, _ in parse_obligation_ranges(
2573
+ file_path.read_text(encoding="utf-8")
2574
+ ):
2575
+ candidates.setdefault(GCD.normalize(obligation.text), []).append(
2576
+ (relative, tuple(obligation.chain))
2577
+ )
2578
+ return candidates
2579
+
2580
+
2581
+ def skeleton(repo: Path, expected: Iterable[ExpectedRow], *, auto_bind_exact: bool) -> str:
2582
+ candidates = exact_candidates(repo) if auto_bind_exact else {}
2583
+ values = []
2584
+ for row in expected:
2585
+ exact = candidates.get(row.before_text, [])
2586
+ mechanically_bound = len(exact) == 1 and exact[0][1] != row.before_chain
2587
+ if mechanically_bound:
2588
+ carrier_path, carrier_chain = exact[0]
2589
+ disposition = "rehosted"
2590
+ effect = "preserved"
2591
+ carrier_text: str | None = row.before_text
2592
+ manual_reviewed = False
2593
+ review_note = "Mechanically preserved verbatim under one exact current (path, chain, text) carrier."
2594
+ else:
2595
+ disposition = "rehosted"
2596
+ effect = "unresolved"
2597
+ carrier_path = None
2598
+ carrier_text = None
2599
+ carrier_chain = None
2600
+ manual_reviewed = False
2601
+ review_note = "Unresolved: exact current carrier and qualifier preservation require review."
2602
+ values.append(
2603
+ json.dumps(
2604
+ {
2605
+ "schema_version": SCHEMA_VERSION,
2606
+ "source_path": row.source_path,
2607
+ "source_ordinal": row.source_ordinal,
2608
+ "reason": row.reason,
2609
+ "before_text": row.before_text,
2610
+ "before_chain": list(row.before_chain),
2611
+ "disposition": disposition,
2612
+ "effect": effect,
2613
+ "carrier_path": carrier_path,
2614
+ "carrier_text": carrier_text,
2615
+ "carrier_chain": list(carrier_chain) if carrier_chain is not None else None,
2616
+ "carrier_bundle": None,
2617
+ "compound_clauses": None,
2618
+ "manual_reviewed": manual_reviewed,
2619
+ "semantic_review": None,
2620
+ "semantic_rationale": None,
2621
+ "qualifiers": [],
2622
+ "qualifier_relations": [],
2623
+ "review_note": review_note,
2624
+ },
2625
+ ensure_ascii=False,
2626
+ separators=(",", ":"),
2627
+ )
2628
+ )
2629
+ return "\n".join(values) + ("\n" if values else "")
2630
+
2631
+
2632
+ def write_atomic(path: Path, content: str) -> None:
2633
+ path.parent.mkdir(parents=True, exist_ok=True)
2634
+ temporary = path.with_name(path.name + ".tmp")
2635
+ temporary.write_text(content, encoding="utf-8")
2636
+ temporary.replace(path)
2637
+
2638
+
2639
+ def main(argv: list[str]) -> int:
2640
+ parser = argparse.ArgumentParser(description=__doc__)
2641
+ parser.add_argument("command", choices=("inventory", "render", "audit"))
2642
+ parser.add_argument("--repo", default=".")
2643
+ parser.add_argument("--base", required=True)
2644
+ parser.add_argument(
2645
+ "--head",
2646
+ help=(
2647
+ "pin the comparison-domain head to this commit; without it the "
2648
+ "domain runs to the working tree, so a repository-frozen ledger "
2649
+ "would demand rows from every later, unrelated change"
2650
+ ),
2651
+ )
2652
+ parser.add_argument("--mapping")
2653
+ parser.add_argument("--ledger")
2654
+ parser.add_argument("--output")
2655
+ parser.add_argument(
2656
+ "--no-auto-bind-exact",
2657
+ action="store_true",
2658
+ help="leave even globally unique verbatim rehosts unresolved in inventory output",
2659
+ )
2660
+ args = parser.parse_args(argv)
2661
+
2662
+ repo = Path(args.repo).resolve()
2663
+ try:
2664
+ # A movable ref (origin/dev) is not a reproducible baseline: resolve it
2665
+ # to a commit ONCE and derive everything — header included — from that
2666
+ # commit, so the rendered header, the comparison domain, and the rows
2667
+ # cannot disagree even if the ref moves mid-run. The header records
2668
+ # ONLY the resolved SHA, so a render invoked via any ref spelling is
2669
+ # byte-identical and audit's byte comparison turns silent baseline
2670
+ # drift into an explicit STALE_LEDGER failure.
2671
+ resolved_base = git(repo, "rev-parse", f"{args.base}^{{commit}}").stdout.strip()
2672
+ resolved_head = (
2673
+ git(repo, "rev-parse", f"{args.head}^{{commit}}").stdout.strip()
2674
+ if args.head
2675
+ else None
2676
+ )
2677
+ comparison = changed_preexisting_paths(repo, resolved_base, resolved_head)
2678
+ expected = derive_rows(repo, resolved_base, comparison, resolved_head)
2679
+ if args.command == "inventory":
2680
+ content = skeleton(repo, expected, auto_bind_exact=not args.no_auto_bind_exact)
2681
+ if args.output:
2682
+ write_atomic(Path(args.output), content)
2683
+ else:
2684
+ sys.stdout.write(content)
2685
+ print(
2686
+ f"inventory_ok domain={len(comparison)} rows={len(expected)}",
2687
+ file=sys.stderr,
2688
+ )
2689
+ return 0
2690
+
2691
+ if not args.mapping:
2692
+ raise AuditError("MAPPING_REQUIRED", "--mapping")
2693
+ mapping_path = Path(args.mapping)
2694
+ mapping_rows = load_mapping(mapping_path)
2695
+ relocations = relocation_destinations(mapping_rows)
2696
+ # Relocation destinations belong to the comparison domain even if they
2697
+ # are new or unchanged; only pre-existing changed paths can owe rows.
2698
+ domain = sorted(set(comparison) | set(relocations))
2699
+ resolved, unresolved = validate_mapping(
2700
+ repo,
2701
+ expected,
2702
+ mapping_rows,
2703
+ allow_unresolved=args.command == "render",
2704
+ )
2705
+ modes = proof_mode_counts(expected, mapping_rows, resolved)
2706
+ rendered = render_ledger(
2707
+ resolved_base,
2708
+ resolved_head,
2709
+ comparison,
2710
+ sorted(set(relocations) - set(comparison)),
2711
+ expected,
2712
+ mapping_rows,
2713
+ resolved,
2714
+ unresolved,
2715
+ )
2716
+ if args.command == "render":
2717
+ if not args.output:
2718
+ sys.stdout.write(rendered)
2719
+ else:
2720
+ write_atomic(Path(args.output), rendered)
2721
+ print(
2722
+ f"render_ok domain={len(domain)} rows={len(expected)} "
2723
+ f"unresolved={len(unresolved)} "
2724
+ f"exact_mechanical={modes['exact-mechanical']} "
2725
+ f"reviewed_semantic={modes['reviewed-semantic']}",
2726
+ file=sys.stderr,
2727
+ )
2728
+ return 0
2729
+
2730
+ if not args.ledger:
2731
+ raise AuditError("LEDGER_REQUIRED", "--ledger")
2732
+ ledger = Path(args.ledger)
2733
+ if not ledger.is_file() or ledger.read_text(encoding="utf-8") != rendered:
2734
+ raise AuditError("STALE_LEDGER", str(ledger))
2735
+ print(
2736
+ f"audit_ok domain={len(domain)} rows={len(expected)} unresolved=0 "
2737
+ f"exact_mechanical={modes['exact-mechanical']} "
2738
+ f"reviewed_semantic={modes['reviewed-semantic']}",
2739
+ file=sys.stderr,
2740
+ )
2741
+ return 0
2742
+ except AuditError as exc:
2743
+ print(f"ERROR {exc.code}: {exc.detail}", file=sys.stderr)
2744
+ return 1
2745
+
2746
+
2747
+ if __name__ == "__main__":
2748
+ raise SystemExit(main(sys.argv[1:]))