@ccoalm/ccl-skills 0.6.2 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/hooks.json +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/remind-unverified-cli-flag.sh +309 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_remind_unverified_cli_flag.sh +483 -0
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/ccl-skills.ts +5 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/SKILL.md +10 -8
- package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/references/mobile-quality-release.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +16 -17
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/client-routing.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +195 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/timeout-auth-and-capabilities.md +3 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/claude_review.sh +13 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh +9 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/kimi_review.sh +9 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/normalize_review_timeout.sh +22 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/opencode_review.sh +9 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +1540 -129
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_claude_review_probe.sh +8 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +76 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +1858 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_update_review_plan_intent.sh +789 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/update_review_plan_intent.py +513 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/architecture-playbook.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/data-platform-architecture.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/event-driven-architecture.md +14 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/multi-tenant-isolation.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/SKILL.md +5 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/SKILL.md +2 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +13 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/SKILL.md +64 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/async-lifecycle-and-performance.md +72 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/runtime-and-project-contract.md +58 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/source-map.md +41 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/verification-diagnostics-and-security.md +63 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/sli-slo-design.md +25 -9
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/source-register.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/promotion-gate-and-review.md +16 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/secret-and-config-management.md +7 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/retry-timeout-circuit-breaker.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +8 -10
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/design-routing-and-readiness.md +10 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/verify-developer-experience.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/SKILL.md +135 -86
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/behavioral-aesthetic-logic.md +66 -80
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/delivery-contract.md +275 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-execution-checklist.md +88 -214
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-impl-naming-and-versioning.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-intake-and-acceptance.md +10 -8
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-system-source-of-truth.md +4 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md +112 -95
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/frontend-code-evidence-map.md +30 -21
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/interaction-design-patterns.md +22 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/layout-recipes-and-screenshot-acceptance.md +20 -17
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/multi-project-token-consistency.md +7 -9
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/multi-stack-strategy.md +14 -10
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/operational-processing-workflows.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/platform-mobile-patterns.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/product-lifecycle-acceptance-and-iteration.md +9 -6
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/product-surface-patterns.md +3 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/source-map.md +37 -10
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/tokens-and-components.md +7 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/ui-ux-audit.md +8 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/ui-ux-design-development.md +16 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/visual-craft.md +4 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/SKILL.md +5 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/architecture-playbook.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/audit-history-architecture.md +31 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/data-platform-architecture.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/event-driven-architecture.md +7 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/multi-tenant-isolation.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/notification-architecture.md +28 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/packaging-runtime-readiness.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/replay-comparison-architecture.md +28 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/workflow-state-architecture.md +39 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/SKILL.md +10 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/ai-service-wiring-patterns.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/audit-history-patterns.md +29 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/background-job-patterns.md +16 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/batch-and-artifact-patterns.md +25 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/notification-patterns.md +40 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/public-api-security-patterns.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/replay-comparison-patterns.md +30 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/state-machine-task-patterns.md +48 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/testing-and-quality-patterns.md +10 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/release-coordination/SKILL.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +4 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/coverage-exhaustion-traps.md +45 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +142 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +21 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +11 -9
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/firing-point-placement.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/parallel-stack-references-pattern.md +5 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/r0-leakage-audit.md +102 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +69 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-to-skill-extraction.md +10 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/uiux-judgment-extraction.md +6 -6
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +4 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-ccl-skills.sh +93 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-parallel-stack-parity.sh +119 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +22 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +49 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/obligation-ledger.py +2748 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/register-firing-path-resolution.rb +20 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/shared_git_surface_gate.py +1142 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_parallel_stack_parity.sh +183 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +19 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_skill_catalog.sh +41 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ci_checkout_ref_binding.sh +120 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_entrypoint_domain_scan_terms.sh +82 -8
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +336 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_self_adjudication.sh +82 -10
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_obligation_ledger.sh +1416 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_obligation_ledger_repo_audit.sh +57 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_wiring.sh +141 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_routing_pointer_integrity.sh +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_shared_git_surface_gate.sh +1696 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_uiux_delivery_contract.sh +2117 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_uiux_loading_budget.sh +316 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +1176 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_skill_cross_refs.sh +31 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate-skill.sh +9 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +980 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/terminal-cli-dev/SKILL.md +9 -6
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/SKILL.md +11 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/client-runtime-test-matrices.md +10 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/fitness-functions.md +16 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/scenario-testing.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/test-code-authoring-patterns.md +16 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/SKILL.md +5 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/delivery-face-closeout.md +16 -6
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/self-benchmark-baseline.md +37 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/SKILL.md +7 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/complex-workspace-patterns.md +1 -1
- package/dist/assets/release.json +275 -105
- package/package.json +1 -1
|
@@ -0,0 +1,2748 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Generate and audit a closed obligation-preservation ledger.
|
|
3
|
+
|
|
4
|
+
The row set is derived from every pre-existing ``skills/**/*.md`` path changed
|
|
5
|
+
against an explicit base revision. Humans bind each derived row to one exact
|
|
6
|
+
current carrier in a JSONL mapping; this program never guesses or fuzzily binds
|
|
7
|
+
carriers. Line ranges in the reader-facing ledger are derived from the current
|
|
8
|
+
files, so stale locators are detectable without putting a dirty-tree digest in
|
|
9
|
+
the generated document.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import argparse
|
|
15
|
+
import hashlib
|
|
16
|
+
import importlib.util
|
|
17
|
+
import json
|
|
18
|
+
import re
|
|
19
|
+
import subprocess
|
|
20
|
+
import sys
|
|
21
|
+
from collections import Counter
|
|
22
|
+
from dataclasses import dataclass
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
from typing import Any, Iterable
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
SCHEMA_VERSION = 3
|
|
28
|
+
SCHEMA_VERSIONS = {3, 4}
|
|
29
|
+
DISPOSITIONS = {
|
|
30
|
+
"merged",
|
|
31
|
+
"subsumed",
|
|
32
|
+
"rehosted",
|
|
33
|
+
"retired-dead",
|
|
34
|
+
"partitioned",
|
|
35
|
+
"partial-retirement",
|
|
36
|
+
}
|
|
37
|
+
EFFECTS = {"preserved", "strengthened", "retired", "unresolved"}
|
|
38
|
+
QUALIFIER_KINDS = {
|
|
39
|
+
"modality",
|
|
40
|
+
"recency",
|
|
41
|
+
"threshold",
|
|
42
|
+
"scope",
|
|
43
|
+
"actor",
|
|
44
|
+
"consequence",
|
|
45
|
+
}
|
|
46
|
+
MAPPING_FIELDS = {
|
|
47
|
+
"schema_version",
|
|
48
|
+
"source_path",
|
|
49
|
+
"source_ordinal",
|
|
50
|
+
"reason",
|
|
51
|
+
"before_text",
|
|
52
|
+
"before_chain",
|
|
53
|
+
"disposition",
|
|
54
|
+
"effect",
|
|
55
|
+
"carrier_path",
|
|
56
|
+
"carrier_text",
|
|
57
|
+
"carrier_chain",
|
|
58
|
+
"carrier_bundle",
|
|
59
|
+
"compound_clauses",
|
|
60
|
+
"manual_reviewed",
|
|
61
|
+
"semantic_review",
|
|
62
|
+
"semantic_rationale",
|
|
63
|
+
"qualifiers",
|
|
64
|
+
"qualifier_relations",
|
|
65
|
+
"review_note",
|
|
66
|
+
"_mapping_line",
|
|
67
|
+
}
|
|
68
|
+
MAPPING_FIELDS_V4 = {
|
|
69
|
+
"schema_version",
|
|
70
|
+
"source_path",
|
|
71
|
+
"source_ordinal",
|
|
72
|
+
"reason",
|
|
73
|
+
"before_text",
|
|
74
|
+
"before_chain",
|
|
75
|
+
"disposition",
|
|
76
|
+
"effect",
|
|
77
|
+
"parts",
|
|
78
|
+
"manual_reviewed",
|
|
79
|
+
"semantic_review",
|
|
80
|
+
"semantic_rationale",
|
|
81
|
+
"review_note",
|
|
82
|
+
"_mapping_line",
|
|
83
|
+
}
|
|
84
|
+
REASONS = {"left-a-rewritten-line", "governing-chain-changed"}
|
|
85
|
+
PROVENANCE_BASENAMES = {"source-register.md", "provenance.md"}
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _load_chain_diff(script_dir: Path) -> Any:
|
|
89
|
+
path = script_dir / "governing-chain-diff.py"
|
|
90
|
+
spec = importlib.util.spec_from_file_location("governing_chain_diff", path)
|
|
91
|
+
if spec is None or spec.loader is None:
|
|
92
|
+
raise RuntimeError(f"cannot import {path}")
|
|
93
|
+
module = importlib.util.module_from_spec(spec)
|
|
94
|
+
sys.modules[spec.name] = module
|
|
95
|
+
spec.loader.exec_module(module)
|
|
96
|
+
return module
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
GCD = _load_chain_diff(Path(__file__).resolve().parent)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
class AuditError(Exception):
|
|
103
|
+
def __init__(self, code: str, detail: str):
|
|
104
|
+
super().__init__(detail)
|
|
105
|
+
self.code = code
|
|
106
|
+
self.detail = detail
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
@dataclass(frozen=True)
|
|
110
|
+
class ExpectedRow:
|
|
111
|
+
source_path: str
|
|
112
|
+
source_ordinal: int
|
|
113
|
+
reason: str
|
|
114
|
+
before_text: str
|
|
115
|
+
before_chain: tuple[str, ...]
|
|
116
|
+
|
|
117
|
+
@property
|
|
118
|
+
def key(self) -> tuple[str, int]:
|
|
119
|
+
return (self.source_path, self.source_ordinal)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
@dataclass(frozen=True)
|
|
123
|
+
class Carrier:
|
|
124
|
+
path: str
|
|
125
|
+
text: str
|
|
126
|
+
chain: tuple[str, ...]
|
|
127
|
+
start_line: int
|
|
128
|
+
end_line: int
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
@dataclass(frozen=True)
|
|
132
|
+
class PartitionedPart:
|
|
133
|
+
part_id: str
|
|
134
|
+
status: str
|
|
135
|
+
source_text: str
|
|
136
|
+
carriers: tuple[Carrier, ...]
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
@dataclass(frozen=True)
|
|
140
|
+
class LedgerObligation:
|
|
141
|
+
text: str
|
|
142
|
+
chain: tuple[str, ...]
|
|
143
|
+
|
|
144
|
+
@property
|
|
145
|
+
def key(self) -> str:
|
|
146
|
+
return GCD.normalize(self.text)
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def git(repo: Path, *args: str, check: bool = True) -> subprocess.CompletedProcess[str]:
|
|
150
|
+
result = subprocess.run(
|
|
151
|
+
["git", "-C", str(repo), *args], capture_output=True, text=True
|
|
152
|
+
)
|
|
153
|
+
if check and result.returncode != 0:
|
|
154
|
+
raise AuditError("GIT_FAILED", result.stderr.strip() or "git command failed")
|
|
155
|
+
return result
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def normalize_path(value: str) -> str:
|
|
159
|
+
path = Path(value)
|
|
160
|
+
if path.is_absolute() or ".." in path.parts:
|
|
161
|
+
raise AuditError("INVALID_PATH", f"path must be repository-relative: {value}")
|
|
162
|
+
return path.as_posix()
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def changed_preexisting_paths(
|
|
166
|
+
repo: Path, base: str, head: str | None = None
|
|
167
|
+
) -> list[str]:
|
|
168
|
+
diff_args = ["diff", "--name-only", "--diff-filter=ACDMRTUXB", base]
|
|
169
|
+
if head is not None:
|
|
170
|
+
diff_args.append(head)
|
|
171
|
+
changed = git(
|
|
172
|
+
repo,
|
|
173
|
+
*diff_args,
|
|
174
|
+
"--",
|
|
175
|
+
"skills",
|
|
176
|
+
).stdout.splitlines()
|
|
177
|
+
out: list[str] = []
|
|
178
|
+
for raw in changed:
|
|
179
|
+
path = normalize_path(raw)
|
|
180
|
+
if not path.startswith("skills/") or not path.endswith(".md"):
|
|
181
|
+
continue
|
|
182
|
+
exists_at_base = git(repo, "cat-file", "-e", f"{base}:{path}", check=False)
|
|
183
|
+
if exists_at_base.returncode == 0:
|
|
184
|
+
out.append(path)
|
|
185
|
+
return sorted(set(out))
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def read_base(repo: Path, base: str, path: str) -> str:
|
|
189
|
+
return git(repo, "show", f"{base}:{path}").stdout
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def read_current(repo: Path, path: str) -> str:
|
|
193
|
+
file_path = repo / path
|
|
194
|
+
return file_path.read_text(encoding="utf-8") if file_path.is_file() else ""
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def read_at(repo: Path, rev: str, path: str) -> str:
|
|
198
|
+
result = git(repo, "show", f"{rev}:{path}", check=False)
|
|
199
|
+
return result.stdout if result.returncode == 0 else ""
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def derive_rows(
|
|
203
|
+
repo: Path, base: str, paths: Iterable[str], head: str | None = None
|
|
204
|
+
) -> list[ExpectedRow]:
|
|
205
|
+
rows: list[ExpectedRow] = []
|
|
206
|
+
for path in sorted(paths):
|
|
207
|
+
before = read_base(repo, base, path)
|
|
208
|
+
after = read_current(repo, path) if head is None else read_at(repo, head, path)
|
|
209
|
+
after_by_key: dict[str, list[Any]] = {}
|
|
210
|
+
for obligation, _, _ in parse_obligation_ranges(after):
|
|
211
|
+
after_by_key.setdefault(obligation.key, []).append(obligation)
|
|
212
|
+
ordinal = 0
|
|
213
|
+
for obligation, _, _ in parse_obligation_ranges(before):
|
|
214
|
+
survivors = after_by_key.get(obligation.key)
|
|
215
|
+
reason: str | None = None
|
|
216
|
+
if not survivors:
|
|
217
|
+
reason = "left-a-rewritten-line"
|
|
218
|
+
elif not any(item.chain == obligation.chain for item in survivors):
|
|
219
|
+
reason = "governing-chain-changed"
|
|
220
|
+
if reason is None:
|
|
221
|
+
continue
|
|
222
|
+
ordinal += 1
|
|
223
|
+
rows.append(
|
|
224
|
+
ExpectedRow(
|
|
225
|
+
source_path=path,
|
|
226
|
+
source_ordinal=ordinal,
|
|
227
|
+
reason=reason,
|
|
228
|
+
before_text=GCD.normalize(obligation.text),
|
|
229
|
+
before_chain=tuple(obligation.chain),
|
|
230
|
+
)
|
|
231
|
+
)
|
|
232
|
+
return rows
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def load_mapping(path: Path) -> list[dict[str, Any]]:
|
|
236
|
+
rows: list[dict[str, Any]] = []
|
|
237
|
+
if not path.is_file():
|
|
238
|
+
raise AuditError("MAPPING_MISSING", str(path))
|
|
239
|
+
for line_no, raw in enumerate(path.read_text(encoding="utf-8").splitlines(), 1):
|
|
240
|
+
if not raw.strip():
|
|
241
|
+
continue
|
|
242
|
+
try:
|
|
243
|
+
value = json.loads(raw)
|
|
244
|
+
except json.JSONDecodeError as exc:
|
|
245
|
+
raise AuditError("MAPPING_JSON", f"{path}:{line_no}: {exc}") from exc
|
|
246
|
+
if not isinstance(value, dict):
|
|
247
|
+
raise AuditError("MAPPING_JSON", f"{path}:{line_no}: object required")
|
|
248
|
+
value["_mapping_line"] = line_no
|
|
249
|
+
rows.append(value)
|
|
250
|
+
return rows
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def mapping_key(row: dict[str, Any]) -> tuple[str, int]:
|
|
254
|
+
try:
|
|
255
|
+
return (normalize_path(str(row["source_path"])), int(row["source_ordinal"]))
|
|
256
|
+
except (KeyError, TypeError, ValueError) as exc:
|
|
257
|
+
raise AuditError("MAPPING_KEY", f"line {row.get('_mapping_line', '?')}: {exc}") from exc
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def index_mapping(rows: list[dict[str, Any]]) -> dict[tuple[str, int], dict[str, Any]]:
|
|
261
|
+
indexed: dict[tuple[str, int], dict[str, Any]] = {}
|
|
262
|
+
for row in rows:
|
|
263
|
+
key = mapping_key(row)
|
|
264
|
+
if key in indexed:
|
|
265
|
+
raise AuditError("DUPLICATE_ROW", f"duplicate mapping key {key[0]}#{key[1]}")
|
|
266
|
+
indexed[key] = row
|
|
267
|
+
return indexed
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def relocation_destinations(rows: Iterable[dict[str, Any]]) -> list[str]:
|
|
271
|
+
paths: set[str] = set()
|
|
272
|
+
for row in rows:
|
|
273
|
+
carrier_path = row.get("carrier_path")
|
|
274
|
+
if carrier_path:
|
|
275
|
+
path = normalize_path(str(carrier_path))
|
|
276
|
+
if not path.startswith("skills/") or not path.endswith(".md"):
|
|
277
|
+
raise AuditError("CARRIER_OUTSIDE_SKILLS", path)
|
|
278
|
+
paths.add(path)
|
|
279
|
+
bundle = row.get("carrier_bundle")
|
|
280
|
+
if isinstance(bundle, list):
|
|
281
|
+
for member in bundle:
|
|
282
|
+
if not isinstance(member, dict) or not member.get("carrier_path"):
|
|
283
|
+
continue
|
|
284
|
+
path = normalize_path(str(member["carrier_path"]))
|
|
285
|
+
if not path.startswith("skills/") or not path.endswith(".md"):
|
|
286
|
+
raise AuditError("CARRIER_OUTSIDE_SKILLS", path)
|
|
287
|
+
paths.add(path)
|
|
288
|
+
parts = row.get("parts")
|
|
289
|
+
if isinstance(parts, list):
|
|
290
|
+
for part in parts:
|
|
291
|
+
if not isinstance(part, dict):
|
|
292
|
+
continue
|
|
293
|
+
carriers = part.get("carriers")
|
|
294
|
+
if not isinstance(carriers, list):
|
|
295
|
+
continue
|
|
296
|
+
for member in carriers:
|
|
297
|
+
if not isinstance(member, dict) or not member.get("carrier_path"):
|
|
298
|
+
continue
|
|
299
|
+
path = normalize_path(str(member["carrier_path"]))
|
|
300
|
+
if not path.startswith("skills/") or not path.endswith(".md"):
|
|
301
|
+
raise AuditError("CARRIER_OUTSIDE_SKILLS", path)
|
|
302
|
+
paths.add(path)
|
|
303
|
+
return sorted(paths)
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
def verify_closed_row_set(
|
|
307
|
+
expected: list[ExpectedRow], mapping: dict[tuple[str, int], dict[str, Any]]
|
|
308
|
+
) -> None:
|
|
309
|
+
expected_by_key = {row.key: row for row in expected}
|
|
310
|
+
missing = sorted(set(expected_by_key) - set(mapping))
|
|
311
|
+
extra = sorted(set(mapping) - set(expected_by_key))
|
|
312
|
+
if missing or extra:
|
|
313
|
+
detail = []
|
|
314
|
+
if missing:
|
|
315
|
+
detail.append("missing=" + ",".join(f"{p}#{n}" for p, n in missing[:12]))
|
|
316
|
+
if extra:
|
|
317
|
+
detail.append("extra=" + ",".join(f"{p}#{n}" for p, n in extra[:12]))
|
|
318
|
+
raise AuditError("ROW_SET_MISMATCH", "; ".join(detail))
|
|
319
|
+
for key, expected_row in expected_by_key.items():
|
|
320
|
+
actual = mapping[key]
|
|
321
|
+
checks = {
|
|
322
|
+
"reason": expected_row.reason,
|
|
323
|
+
"before_text": expected_row.before_text,
|
|
324
|
+
"before_chain": list(expected_row.before_chain),
|
|
325
|
+
}
|
|
326
|
+
for field, wanted in checks.items():
|
|
327
|
+
if actual.get(field) != wanted:
|
|
328
|
+
raise AuditError(
|
|
329
|
+
"ROW_SOURCE_MISMATCH",
|
|
330
|
+
f"{key[0]}#{key[1]} {field}: expected {wanted!r}, got {actual.get(field)!r}",
|
|
331
|
+
)
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def normalized_with_offsets(source: str) -> tuple[str, list[int]]:
|
|
335
|
+
chars: list[str] = []
|
|
336
|
+
offsets: list[int] = []
|
|
337
|
+
in_space = False
|
|
338
|
+
for index, char in enumerate(source):
|
|
339
|
+
if char.isspace():
|
|
340
|
+
if chars and not in_space:
|
|
341
|
+
chars.append(" ")
|
|
342
|
+
offsets.append(index)
|
|
343
|
+
in_space = True
|
|
344
|
+
else:
|
|
345
|
+
chars.append(char)
|
|
346
|
+
offsets.append(index)
|
|
347
|
+
in_space = False
|
|
348
|
+
if chars and chars[-1] == " ":
|
|
349
|
+
chars.pop()
|
|
350
|
+
offsets.pop()
|
|
351
|
+
return "".join(chars), offsets
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
def line_range(source: str, start: int, end: int) -> tuple[int, int]:
|
|
355
|
+
return (source.count("\n", 0, start) + 1, source.count("\n", 0, end) + 1)
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def _blank(chars: list[str], start: int, end: int) -> None:
|
|
359
|
+
for index in range(start, end):
|
|
360
|
+
if chars[index] not in "\r\n":
|
|
361
|
+
chars[index] = " "
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
_FENCE_OPEN = re.compile(r"^[ ]{0,3}(`{3,}|~{3,})([^\r\n]*)$")
|
|
365
|
+
_ATX_HEADING = re.compile(r"^[ ]{0,3}#{1,6}(?:[ \t]|$)")
|
|
366
|
+
_THEMATIC_BREAK = re.compile(r"^[ ]{0,3}([-_*])[ \t]*(?:\1[ \t]*){2,}$")
|
|
367
|
+
_SETEXT_UNDERLINE = re.compile(r"^[ ]{0,3}(?:=+|-+)[ \t]*$")
|
|
368
|
+
_BLOCK_QUOTE_PREFIX = re.compile(r"^[ ]{0,3}(?:>[ ]?)+")
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
def mask_inert_markdown(source: str) -> str:
|
|
372
|
+
"""Blank comments and code containers without changing source offsets.
|
|
373
|
+
|
|
374
|
+
Fence closing follows the relevant CommonMark invariant: same marker,
|
|
375
|
+
closing run at least as long as the opener, and whitespace-only suffix.
|
|
376
|
+
Indented code is also inert — at top level from column 4, inside a list
|
|
377
|
+
item from the content column + 4 — but never where the indented line
|
|
378
|
+
lazily continues an open paragraph; those list continuations remain
|
|
379
|
+
visible to the prose parser.
|
|
380
|
+
"""
|
|
381
|
+
chars = list(source)
|
|
382
|
+
comment_start = 0
|
|
383
|
+
while True:
|
|
384
|
+
comment_start = source.find("<!--", comment_start)
|
|
385
|
+
if comment_start < 0:
|
|
386
|
+
break
|
|
387
|
+
comment_end = source.find("-->", comment_start + 4)
|
|
388
|
+
comment_end = len(source) if comment_end < 0 else comment_end + 3
|
|
389
|
+
_blank(chars, comment_start, comment_end)
|
|
390
|
+
comment_start = comment_end
|
|
391
|
+
|
|
392
|
+
commentless = "".join(chars)
|
|
393
|
+
lines = commentless.splitlines(keepends=True)
|
|
394
|
+
offset = 0
|
|
395
|
+
fence_marker: str | None = None
|
|
396
|
+
fence_length = 0
|
|
397
|
+
list_content_column: int | None = None
|
|
398
|
+
paragraph_open = False
|
|
399
|
+
for raw_with_end in lines:
|
|
400
|
+
raw = raw_with_end.rstrip("\r\n")
|
|
401
|
+
line_start = offset
|
|
402
|
+
offset += len(raw_with_end)
|
|
403
|
+
|
|
404
|
+
if fence_marker is not None:
|
|
405
|
+
closer = re.match(
|
|
406
|
+
rf"^[ ]{{0,3}}({re.escape(fence_marker)}{{{fence_length},}})[ \t]*$",
|
|
407
|
+
raw,
|
|
408
|
+
)
|
|
409
|
+
_blank(chars, line_start, offset)
|
|
410
|
+
if closer:
|
|
411
|
+
fence_marker = None
|
|
412
|
+
fence_length = 0
|
|
413
|
+
paragraph_open = False
|
|
414
|
+
continue
|
|
415
|
+
|
|
416
|
+
opener = _FENCE_OPEN.match(raw)
|
|
417
|
+
if opener and not (
|
|
418
|
+
opener.group(1).startswith("`") and "`" in opener.group(2)
|
|
419
|
+
):
|
|
420
|
+
fence_marker = opener.group(1)[0]
|
|
421
|
+
fence_length = len(opener.group(1))
|
|
422
|
+
_blank(chars, line_start, offset)
|
|
423
|
+
paragraph_open = False
|
|
424
|
+
continue
|
|
425
|
+
|
|
426
|
+
if not raw.strip():
|
|
427
|
+
paragraph_open = False
|
|
428
|
+
continue
|
|
429
|
+
item = GCD._LIST_ITEM.match(raw)
|
|
430
|
+
if item:
|
|
431
|
+
list_content_column = len(raw[: item.start(2)].expandtabs(4))
|
|
432
|
+
# The item's FIRST content line decides the paragraph state: a
|
|
433
|
+
# heading or thematic break there does not open a paragraph, so
|
|
434
|
+
# content-column + 4 indentation after it is code.
|
|
435
|
+
item_text = raw[item.start(2) :]
|
|
436
|
+
paragraph_open = bool(item_text.strip()) and not (
|
|
437
|
+
_ATX_HEADING.match(item_text) or _THEMATIC_BREAK.match(item_text)
|
|
438
|
+
)
|
|
439
|
+
continue
|
|
440
|
+
leading = len(raw) - len(raw.lstrip(" \t"))
|
|
441
|
+
expanded_leading = len(raw[:leading].expandtabs(4))
|
|
442
|
+
if list_content_column is not None and expanded_leading >= list_content_column:
|
|
443
|
+
if expanded_leading - list_content_column >= 4 and not paragraph_open:
|
|
444
|
+
# Indented code inside a list item starts at the content
|
|
445
|
+
# column + 4 and, as at top level, cannot interrupt the
|
|
446
|
+
# item's open paragraph.
|
|
447
|
+
_blank(chars, line_start, offset)
|
|
448
|
+
continue
|
|
449
|
+
content = raw.lstrip(" \t")
|
|
450
|
+
if paragraph_open and _SETEXT_UNDERLINE.match(content):
|
|
451
|
+
paragraph_open = False
|
|
452
|
+
else:
|
|
453
|
+
paragraph_open = not (
|
|
454
|
+
_ATX_HEADING.match(content) or _THEMATIC_BREAK.match(content)
|
|
455
|
+
)
|
|
456
|
+
continue
|
|
457
|
+
if expanded_leading == 0:
|
|
458
|
+
list_content_column = None
|
|
459
|
+
if expanded_leading >= 4 and not paragraph_open:
|
|
460
|
+
# Indented code cannot interrupt a paragraph per CommonMark, but
|
|
461
|
+
# it can start after any other block: blank line, heading, or a
|
|
462
|
+
# closed fence. Continuation lines keep the block open because a
|
|
463
|
+
# masked code line never opens a paragraph.
|
|
464
|
+
_blank(chars, line_start, offset)
|
|
465
|
+
continue
|
|
466
|
+
# Classify the line's block-level content, not its raw prefix: inside
|
|
467
|
+
# a block quote the paragraph state follows the quoted content, so a
|
|
468
|
+
# quoted heading closes paragraph state instead of opening it (while
|
|
469
|
+
# quoted prose still opens it — indented lines after quoted prose are
|
|
470
|
+
# lazy continuations and stay visible).
|
|
471
|
+
content = raw
|
|
472
|
+
quote = _BLOCK_QUOTE_PREFIX.match(content)
|
|
473
|
+
if quote:
|
|
474
|
+
content = content[quote.end() :]
|
|
475
|
+
content_leading = len(content) - len(content.lstrip(" \t"))
|
|
476
|
+
if (
|
|
477
|
+
len(content[:content_leading].expandtabs(4)) >= 4
|
|
478
|
+
and not paragraph_open
|
|
479
|
+
):
|
|
480
|
+
# Indented code inside a block quote is measured from the
|
|
481
|
+
# quote marker, not the raw line start; the same
|
|
482
|
+
# cannot-interrupt-a-paragraph rule applies.
|
|
483
|
+
_blank(chars, line_start, offset)
|
|
484
|
+
continue
|
|
485
|
+
if paragraph_open and _SETEXT_UNDERLINE.match(content):
|
|
486
|
+
# The line closes an open paragraph as a Setext heading underline,
|
|
487
|
+
# so what follows may open an indented code block.
|
|
488
|
+
paragraph_open = False
|
|
489
|
+
else:
|
|
490
|
+
paragraph_open = bool(content.strip()) and not (
|
|
491
|
+
_ATX_HEADING.match(content) or _THEMATIC_BREAK.match(content)
|
|
492
|
+
)
|
|
493
|
+
return "".join(chars)
|
|
494
|
+
|
|
495
|
+
|
|
496
|
+
def inline_code_only(text: str) -> bool:
|
|
497
|
+
normalized = GCD.normalize(text)
|
|
498
|
+
if not normalized.startswith("`"):
|
|
499
|
+
return False
|
|
500
|
+
opener = len(normalized) - len(normalized.lstrip("`"))
|
|
501
|
+
closer = len(normalized) - len(normalized.rstrip("`"))
|
|
502
|
+
if opener != closer or len(normalized) <= opener + closer:
|
|
503
|
+
return False
|
|
504
|
+
inner = normalized[opener : len(normalized) - closer]
|
|
505
|
+
return bool(inner.strip()) and ("`" * opener) not in inner
|
|
506
|
+
|
|
507
|
+
|
|
508
|
+
def prose_obligation_ranges(source: str) -> list[tuple[LedgerObligation, int, int]]:
|
|
509
|
+
"""Recover ranges for prose obligations while excluding HTML comments.
|
|
510
|
+
|
|
511
|
+
Both walks are document ordered. Matching from the previous end makes a
|
|
512
|
+
repeated sentence under two different hosts addressable by the manifest's
|
|
513
|
+
(path, chain, exact text) identity instead of imposing corpus-wide textual
|
|
514
|
+
uniqueness.
|
|
515
|
+
"""
|
|
516
|
+
visible = mask_inert_markdown(source)
|
|
517
|
+
haystack, offsets = normalized_with_offsets(visible)
|
|
518
|
+
cursor = 0
|
|
519
|
+
out: list[tuple[LedgerObligation, int, int]] = []
|
|
520
|
+
for parsed in GCD.parse(visible):
|
|
521
|
+
obligation = LedgerObligation(parsed.text, tuple(parsed.chain))
|
|
522
|
+
if inline_code_only(obligation.text):
|
|
523
|
+
continue
|
|
524
|
+
needle = GCD.normalize(obligation.text)
|
|
525
|
+
pos = haystack.find(needle, cursor)
|
|
526
|
+
if pos < 0:
|
|
527
|
+
raise AuditError(
|
|
528
|
+
"CARRIER_RANGE_UNRECOVERABLE",
|
|
529
|
+
f"cannot locate parsed obligation after normalized offset {cursor}: {needle[:80]}",
|
|
530
|
+
)
|
|
531
|
+
end_pos = pos + len(needle) - 1
|
|
532
|
+
out.append((obligation, offsets[pos], offsets[end_pos] + 1))
|
|
533
|
+
cursor = pos + len(needle)
|
|
534
|
+
return out
|
|
535
|
+
|
|
536
|
+
|
|
537
|
+
_TABLE_SEPARATOR_CELL = re.compile(r"^:?-{3,}:?$")
|
|
538
|
+
|
|
539
|
+
|
|
540
|
+
def code_span_mask(text: str) -> list[bool]:
|
|
541
|
+
"""Mark characters inside matched inline code spans (CommonMark pairing).
|
|
542
|
+
|
|
543
|
+
A run of N backticks opens a span that closes only at the next run of
|
|
544
|
+
exactly N backticks; runs of a different length inside an open span are
|
|
545
|
+
literal, and an unmatched opener is literal text, not an open-forever
|
|
546
|
+
span. Backslash-escaped backticks outside spans are literal.
|
|
547
|
+
"""
|
|
548
|
+
mask = [False] * len(text)
|
|
549
|
+
index = 0
|
|
550
|
+
while index < len(text):
|
|
551
|
+
char = text[index]
|
|
552
|
+
if char == "\\" and index + 1 < len(text):
|
|
553
|
+
index += 2
|
|
554
|
+
continue
|
|
555
|
+
if char != "`":
|
|
556
|
+
index += 1
|
|
557
|
+
continue
|
|
558
|
+
run = 1
|
|
559
|
+
while index + run < len(text) and text[index + run] == "`":
|
|
560
|
+
run += 1
|
|
561
|
+
search = index + run
|
|
562
|
+
closer = -1
|
|
563
|
+
while search < len(text):
|
|
564
|
+
if text[search] != "`":
|
|
565
|
+
search += 1
|
|
566
|
+
continue
|
|
567
|
+
closer_run = 1
|
|
568
|
+
while search + closer_run < len(text) and text[search + closer_run] == "`":
|
|
569
|
+
closer_run += 1
|
|
570
|
+
if closer_run == run:
|
|
571
|
+
closer = search
|
|
572
|
+
break
|
|
573
|
+
search += closer_run
|
|
574
|
+
if closer < 0:
|
|
575
|
+
index += run
|
|
576
|
+
continue
|
|
577
|
+
for position in range(index, closer + run):
|
|
578
|
+
mask[position] = True
|
|
579
|
+
index = closer + run
|
|
580
|
+
return mask
|
|
581
|
+
|
|
582
|
+
|
|
583
|
+
def table_cells(line: str) -> list[tuple[str, int, int]] | None:
|
|
584
|
+
"""Split a pipe table without splitting escaped pipes or inline code."""
|
|
585
|
+
stripped = line.strip()
|
|
586
|
+
if not stripped.startswith("|") or not stripped.endswith("|"):
|
|
587
|
+
return None
|
|
588
|
+
leading = len(line) - len(line.lstrip())
|
|
589
|
+
content = stripped[1:-1]
|
|
590
|
+
in_code = code_span_mask(content)
|
|
591
|
+
cells: list[tuple[str, int, int]] = []
|
|
592
|
+
|
|
593
|
+
def emit(start: int, end: int) -> None:
|
|
594
|
+
raw_cell = content[start:end]
|
|
595
|
+
left_trim = len(raw_cell) - len(raw_cell.lstrip())
|
|
596
|
+
right_trimmed = raw_cell.rstrip()
|
|
597
|
+
cell_start = leading + 1 + start + left_trim
|
|
598
|
+
cell_end = leading + 1 + start + len(right_trimmed)
|
|
599
|
+
cells.append((GCD.normalize(raw_cell), cell_start, cell_end))
|
|
600
|
+
|
|
601
|
+
start = 0
|
|
602
|
+
escaped = False
|
|
603
|
+
for index, char in enumerate(content):
|
|
604
|
+
if escaped:
|
|
605
|
+
escaped = False
|
|
606
|
+
continue
|
|
607
|
+
if in_code[index]:
|
|
608
|
+
continue
|
|
609
|
+
if char == "\\":
|
|
610
|
+
escaped = True
|
|
611
|
+
continue
|
|
612
|
+
if char == "|":
|
|
613
|
+
emit(start, index)
|
|
614
|
+
start = index + 1
|
|
615
|
+
emit(start, len(content))
|
|
616
|
+
return cells
|
|
617
|
+
|
|
618
|
+
|
|
619
|
+
def is_table_separator(line: str) -> bool:
|
|
620
|
+
cells = table_cells(line)
|
|
621
|
+
return bool(cells) and all(
|
|
622
|
+
_TABLE_SEPARATOR_CELL.fullmatch(cell[0]) for cell in cells
|
|
623
|
+
)
|
|
624
|
+
|
|
625
|
+
|
|
626
|
+
def table_obligation_ranges(
|
|
627
|
+
source: str, *, include_inline_code_only: bool = False
|
|
628
|
+
) -> list[tuple[LedgerObligation, int, int]]:
|
|
629
|
+
"""Parse Markdown table data rows as obligations.
|
|
630
|
+
|
|
631
|
+
The header is table identity and therefore an immediate governing host.
|
|
632
|
+
Header/separator rows are schema, not obligations. Complete data-row text
|
|
633
|
+
preserves column/cell semantics and supplies an exact carrier locator.
|
|
634
|
+
"""
|
|
635
|
+
out: list[tuple[LedgerObligation, int, int]] = []
|
|
636
|
+
headings: list[tuple[int, str]] = []
|
|
637
|
+
visible_source = mask_inert_markdown(source)
|
|
638
|
+
lines = visible_source.splitlines(keepends=True)
|
|
639
|
+
offsets: list[int] = []
|
|
640
|
+
total = 0
|
|
641
|
+
for line in lines:
|
|
642
|
+
offsets.append(total)
|
|
643
|
+
total += len(line)
|
|
644
|
+
|
|
645
|
+
table_header: list[str] | None = None
|
|
646
|
+
table_active = False
|
|
647
|
+
for index, raw_with_end in enumerate(lines):
|
|
648
|
+
visible = raw_with_end.rstrip("\r\n")
|
|
649
|
+
|
|
650
|
+
if not visible.strip():
|
|
651
|
+
table_header = None
|
|
652
|
+
table_active = False
|
|
653
|
+
continue
|
|
654
|
+
|
|
655
|
+
heading = GCD._HEADING.match(visible)
|
|
656
|
+
if heading:
|
|
657
|
+
level = len(heading.group(1))
|
|
658
|
+
while headings and headings[-1][0] >= level:
|
|
659
|
+
headings.pop()
|
|
660
|
+
headings.append((level, GCD.content_key(heading.group(2))))
|
|
661
|
+
table_header = None
|
|
662
|
+
table_active = False
|
|
663
|
+
continue
|
|
664
|
+
|
|
665
|
+
cells = table_cells(visible)
|
|
666
|
+
if cells is None:
|
|
667
|
+
table_header = None
|
|
668
|
+
table_active = False
|
|
669
|
+
continue
|
|
670
|
+
if is_table_separator(visible):
|
|
671
|
+
table_active = table_header is not None
|
|
672
|
+
continue
|
|
673
|
+
if not table_active:
|
|
674
|
+
table_header = [cell[0] for cell in cells]
|
|
675
|
+
continue
|
|
676
|
+
|
|
677
|
+
table_identity = " | ".join(table_header or [])
|
|
678
|
+
for column, (text, cell_start, cell_end) in enumerate(cells):
|
|
679
|
+
if not text or (inline_code_only(text) and not include_inline_code_only):
|
|
680
|
+
continue
|
|
681
|
+
header = (
|
|
682
|
+
table_header[column]
|
|
683
|
+
if table_header is not None and column < len(table_header)
|
|
684
|
+
else f"column-{column + 1}"
|
|
685
|
+
)
|
|
686
|
+
chain = tuple(key for _, key in headings) + (
|
|
687
|
+
"table:" + GCD.content_key(table_identity),
|
|
688
|
+
"column:" + GCD.content_key(header),
|
|
689
|
+
)
|
|
690
|
+
start = offsets[index] + cell_start
|
|
691
|
+
end = offsets[index] + cell_end
|
|
692
|
+
out.append((LedgerObligation(text, chain), start, end))
|
|
693
|
+
return out
|
|
694
|
+
|
|
695
|
+
|
|
696
|
+
def parse_obligation_ranges(source: str) -> list[tuple[LedgerObligation, int, int]]:
|
|
697
|
+
obligations = prose_obligation_ranges(source) + table_obligation_ranges(source)
|
|
698
|
+
return sorted(obligations, key=lambda item: (item[1], item[2]))
|
|
699
|
+
|
|
700
|
+
|
|
701
|
+
def raw_list_item_ranges(source: str) -> list[tuple[LedgerObligation, int, int]]:
|
|
702
|
+
"""Expose an exact visible list item as a schema4 carrier without changing row derivation.
|
|
703
|
+
|
|
704
|
+
The legacy row splitter intentionally breaks label-plus-rule bullets into
|
|
705
|
+
clause-sized obligations. A schema4 bundle sometimes needs the complete
|
|
706
|
+
current item (label and operative clauses together). This locator is
|
|
707
|
+
limited to visible Markdown list items; comments, fences, and indented code
|
|
708
|
+
were already blanked by ``mask_inert_markdown``.
|
|
709
|
+
"""
|
|
710
|
+
visible = mask_inert_markdown(source)
|
|
711
|
+
headings: list[tuple[int, str]] = []
|
|
712
|
+
list_stack: list[tuple[int, str]] = []
|
|
713
|
+
out: list[tuple[LedgerObligation, int, int]] = []
|
|
714
|
+
offset = 0
|
|
715
|
+
for raw in visible.splitlines(keepends=True):
|
|
716
|
+
line = raw.rstrip("\r\n")
|
|
717
|
+
heading = GCD._HEADING.match(line)
|
|
718
|
+
if heading:
|
|
719
|
+
level = len(heading.group(1))
|
|
720
|
+
while headings and headings[-1][0] >= level:
|
|
721
|
+
headings.pop()
|
|
722
|
+
headings.append((level, GCD.content_key(heading.group(2))))
|
|
723
|
+
list_stack.clear()
|
|
724
|
+
offset += len(raw)
|
|
725
|
+
continue
|
|
726
|
+
item = GCD._LIST_ITEM.match(line)
|
|
727
|
+
if item:
|
|
728
|
+
indent = len(item.group(1).expandtabs(4))
|
|
729
|
+
while list_stack and list_stack[-1][0] >= indent:
|
|
730
|
+
list_stack.pop()
|
|
731
|
+
text = GCD.normalize(item.group(2))
|
|
732
|
+
if text:
|
|
733
|
+
start = offset + item.start(2)
|
|
734
|
+
end = offset + item.end(2)
|
|
735
|
+
chain = tuple([value for _, value in headings] + [value for _, value in list_stack])
|
|
736
|
+
out.append((LedgerObligation(text, chain), start, end))
|
|
737
|
+
list_stack.append((indent, GCD.content_key(text)))
|
|
738
|
+
elif line and not line[0].isspace():
|
|
739
|
+
list_stack.clear()
|
|
740
|
+
offset += len(raw)
|
|
741
|
+
return out
|
|
742
|
+
|
|
743
|
+
|
|
744
|
+
def raw_table_row_ranges(source: str) -> list[tuple[LedgerObligation, int, int]]:
|
|
745
|
+
"""Expose a complete visible data row, never an inline-code-only cell."""
|
|
746
|
+
visible = mask_inert_markdown(source)
|
|
747
|
+
headings: list[tuple[int, str]] = []
|
|
748
|
+
header: list[str] | None = None
|
|
749
|
+
active = False
|
|
750
|
+
out: list[tuple[LedgerObligation, int, int]] = []
|
|
751
|
+
offset = 0
|
|
752
|
+
for raw in visible.splitlines(keepends=True):
|
|
753
|
+
line = raw.rstrip("\r\n")
|
|
754
|
+
heading = GCD._HEADING.match(line)
|
|
755
|
+
if heading:
|
|
756
|
+
level = len(heading.group(1))
|
|
757
|
+
while headings and headings[-1][0] >= level:
|
|
758
|
+
headings.pop()
|
|
759
|
+
headings.append((level, GCD.content_key(heading.group(2))))
|
|
760
|
+
header = None
|
|
761
|
+
active = False
|
|
762
|
+
offset += len(raw)
|
|
763
|
+
continue
|
|
764
|
+
cells = table_cells(line)
|
|
765
|
+
if cells is None:
|
|
766
|
+
header = None
|
|
767
|
+
active = False
|
|
768
|
+
offset += len(raw)
|
|
769
|
+
continue
|
|
770
|
+
if is_table_separator(line):
|
|
771
|
+
active = header is not None
|
|
772
|
+
offset += len(raw)
|
|
773
|
+
continue
|
|
774
|
+
if not active:
|
|
775
|
+
header = [cell[0] for cell in cells]
|
|
776
|
+
offset += len(raw)
|
|
777
|
+
continue
|
|
778
|
+
text = " | ".join(cell[0] for cell in cells)
|
|
779
|
+
if text:
|
|
780
|
+
identity = " | ".join(header or [])
|
|
781
|
+
chain = tuple(value for _, value in headings) + (
|
|
782
|
+
"table:" + GCD.content_key(identity),
|
|
783
|
+
"row",
|
|
784
|
+
)
|
|
785
|
+
out.append(
|
|
786
|
+
(
|
|
787
|
+
LedgerObligation(text, chain),
|
|
788
|
+
offset + cells[0][1],
|
|
789
|
+
offset + cells[-1][2],
|
|
790
|
+
)
|
|
791
|
+
)
|
|
792
|
+
offset += len(raw)
|
|
793
|
+
return out
|
|
794
|
+
|
|
795
|
+
|
|
796
|
+
def section_body_ranges(source: str) -> list[tuple[LedgerObligation, int, int]]:
|
|
797
|
+
"""Return exact visible section bodies for schema4 closed-bundle carriers."""
|
|
798
|
+
visible = mask_inert_markdown(source)
|
|
799
|
+
lines = visible.splitlines(keepends=True)
|
|
800
|
+
offsets: list[int] = []
|
|
801
|
+
total = 0
|
|
802
|
+
for line in lines:
|
|
803
|
+
offsets.append(total)
|
|
804
|
+
total += len(line)
|
|
805
|
+
headings: list[tuple[int, str]] = []
|
|
806
|
+
found: list[tuple[int, int, tuple[str, ...]]] = []
|
|
807
|
+
for index, raw in enumerate(lines):
|
|
808
|
+
heading = GCD._HEADING.match(raw.rstrip("\r\n"))
|
|
809
|
+
if not heading:
|
|
810
|
+
continue
|
|
811
|
+
level = len(heading.group(1))
|
|
812
|
+
while headings and headings[-1][0] >= level:
|
|
813
|
+
headings.pop()
|
|
814
|
+
chain = tuple([value for _, value in headings] + [GCD.content_key(heading.group(2))])
|
|
815
|
+
found.append((index, level, chain))
|
|
816
|
+
headings.append((level, GCD.content_key(heading.group(2))))
|
|
817
|
+
out: list[tuple[LedgerObligation, int, int]] = []
|
|
818
|
+
for position, (line_index, level, chain) in enumerate(found):
|
|
819
|
+
next_index = len(lines)
|
|
820
|
+
for candidate_index, candidate_level, _ in found[position + 1 :]:
|
|
821
|
+
if candidate_level <= level:
|
|
822
|
+
next_index = candidate_index
|
|
823
|
+
break
|
|
824
|
+
start = offsets[line_index] + len(lines[line_index])
|
|
825
|
+
end = offsets[next_index] if next_index < len(lines) else len(visible)
|
|
826
|
+
while start < end and visible[start].isspace():
|
|
827
|
+
start += 1
|
|
828
|
+
while end > start and visible[end - 1].isspace():
|
|
829
|
+
end -= 1
|
|
830
|
+
text = GCD.normalize(visible[start:end])
|
|
831
|
+
if text:
|
|
832
|
+
out.append((LedgerObligation(text, chain), start, end))
|
|
833
|
+
return out
|
|
834
|
+
|
|
835
|
+
|
|
836
|
+
def exact_carrier(repo: Path, path: str, text: str, chain: list[str]) -> Carrier:
|
|
837
|
+
file_path = repo / path
|
|
838
|
+
if not file_path.is_file():
|
|
839
|
+
raise AuditError("CARRIER_PATH_MISSING", path)
|
|
840
|
+
source = file_path.read_text(encoding="utf-8")
|
|
841
|
+
normalized_text = GCD.normalize(text)
|
|
842
|
+
claimed_chain = tuple(chain)
|
|
843
|
+
exact_text_obligations = [
|
|
844
|
+
(obligation, start, end)
|
|
845
|
+
for obligation, start, end in parse_obligation_ranges(source)
|
|
846
|
+
if GCD.normalize(obligation.text) == normalized_text
|
|
847
|
+
]
|
|
848
|
+
if not exact_text_obligations:
|
|
849
|
+
exact_text_obligations = [
|
|
850
|
+
(obligation, start, end)
|
|
851
|
+
for obligation, start, end in (
|
|
852
|
+
raw_list_item_ranges(source)
|
|
853
|
+
+ raw_table_row_ranges(source)
|
|
854
|
+
+ section_body_ranges(source)
|
|
855
|
+
)
|
|
856
|
+
if GCD.normalize(obligation.text) == normalized_text
|
|
857
|
+
]
|
|
858
|
+
matching_obligations = [
|
|
859
|
+
item for item in exact_text_obligations if tuple(item[0].chain) == claimed_chain
|
|
860
|
+
]
|
|
861
|
+
if not matching_obligations and exact_text_obligations:
|
|
862
|
+
raise AuditError(
|
|
863
|
+
"CARRIER_CHAIN_MISMATCH",
|
|
864
|
+
f"{path}: exact text exists, but not under claimed chain {list(claimed_chain)!r}",
|
|
865
|
+
)
|
|
866
|
+
if len(matching_obligations) != 1:
|
|
867
|
+
raise AuditError(
|
|
868
|
+
"CARRIER_COMPOSITE_NOT_UNIQUE",
|
|
869
|
+
f"{path}: (chain, exact text) count={len(matching_obligations)}",
|
|
870
|
+
)
|
|
871
|
+
obligation, start, end = matching_obligations[0]
|
|
872
|
+
derived_chain = tuple(obligation.chain)
|
|
873
|
+
start_line, end_line = line_range(source, start, end)
|
|
874
|
+
return Carrier(path, normalized_text, derived_chain, start_line, end_line)
|
|
875
|
+
|
|
876
|
+
|
|
877
|
+
QUALIFIER_PATTERNS: dict[str, re.Pattern[str]] = {
|
|
878
|
+
"modality": re.compile(
|
|
879
|
+
r"\b(?:must|required|shall|never|cannot|can't|do not|may|should)\b|必须|不得|禁止|不可|不能|应当|应该|可以",
|
|
880
|
+
re.I,
|
|
881
|
+
),
|
|
882
|
+
"recency": re.compile(
|
|
883
|
+
r"\b(?:before|after|current|latest|final|again|first|initial|prior|subsequent)\b|之前|之后|当前|最新|最终|再次|首次|先|再",
|
|
884
|
+
re.I,
|
|
885
|
+
),
|
|
886
|
+
"scope": re.compile(
|
|
887
|
+
r"\b(?:not\s+a\s+substitute|limited\s+to|restricted\s+to|every|each|all|any|only|solely|exclusively|full|whole|per|none)\b|每个|每条|全部|所有|任何|仅|完整|全量|无一",
|
|
888
|
+
re.I,
|
|
889
|
+
),
|
|
890
|
+
"actor": re.compile(
|
|
891
|
+
r"\b(?:owners?|users?|reviewers?|testers?|testing|clients?|maintainers?|designers?|developers?|agents?)\b|用户|负责人|维护者|评审者|设计师|开发者|测试人员|测试者|客户端|智能体",
|
|
892
|
+
re.I,
|
|
893
|
+
),
|
|
894
|
+
"consequence": re.compile(
|
|
895
|
+
r"\b(?:blocks?|blocked|fails?|failed|stops?|stopped|invalid|rejects?|rejected|pending|cannot claim|must not claim|red)\b|阻塞|失败|停止|无效|拒绝|待定|不得声称|不可声明|红灯",
|
|
896
|
+
re.I,
|
|
897
|
+
),
|
|
898
|
+
}
|
|
899
|
+
|
|
900
|
+
HARD_MODALITY = re.compile(
|
|
901
|
+
r"\b(?:must|required|shall|never|cannot|can't|do not)\b|必须|不得|禁止|不可|不能",
|
|
902
|
+
re.I,
|
|
903
|
+
)
|
|
904
|
+
SOFT_MODALITY = re.compile(r"\b(?:may|should|can)\b|可以|应该|应当", re.I)
|
|
905
|
+
|
|
906
|
+
_THRESHOLD_COMPARATOR = re.compile(
|
|
907
|
+
r"\b(?:at least|at most|minimum|maximum|more than|less than)\b"
|
|
908
|
+
r"|至少|至多|最多|最少|不超过|不少于|高于|低于",
|
|
909
|
+
re.I,
|
|
910
|
+
)
|
|
911
|
+
_THRESHOLD_DIRECTIONAL_COMPARATOR = re.compile(
|
|
912
|
+
r"\b(?:above|below|over|under)\b(?=\s+\d+(?:\.\d+)?)",
|
|
913
|
+
re.I,
|
|
914
|
+
)
|
|
915
|
+
_THRESHOLD_WITHIN = re.compile(
|
|
916
|
+
r"\bwithin\b(?=\s+(?:\d+(?:\.\d+)?|one|two|three|four|five|six|seven|eight|nine|ten)\s*(?:ms|msec(?:ond)?s?|s|sec(?:ond)?s?|min(?:ute)?s?|h|hours?|days?|frames?|attempts?|retries?|requests?)\b)",
|
|
917
|
+
re.I,
|
|
918
|
+
)
|
|
919
|
+
_THRESHOLD_RATIO = re.compile(
|
|
920
|
+
r"(?<![A-Za-z0-9_])\d+(?:\.\d+)?\s*:\s*\d+(?:\.\d+)?(?![A-Za-z0-9_])"
|
|
921
|
+
)
|
|
922
|
+
_THRESHOLD_DIMENSION = re.compile(
|
|
923
|
+
r"(?<![A-Za-z0-9_])"
|
|
924
|
+
r"\d+(?:\.\d+)?\s*[x×]\s*\d+(?:\.\d+)?"
|
|
925
|
+
r"(?:\s*(?:px|pt|dp|sp))?"
|
|
926
|
+
r"(?![A-Za-z0-9_])",
|
|
927
|
+
re.I,
|
|
928
|
+
)
|
|
929
|
+
_THRESHOLD_QUANTITY = re.compile(
|
|
930
|
+
r"(?<![A-Za-z0-9_])"
|
|
931
|
+
r"\d+(?:\.\d+)?\s*"
|
|
932
|
+
r"(?:%|px|pt|dp|sp|ms|msec(?:ond)?s?|s|sec(?:ond)?s?|min(?:ute)?s?|h|hours?|days?|bytes?|kib|mib|gib|kb|mb|gb|items?|rows?|files?|steps?|times?|characters?|chars?|screens?|locales?|variants?|states?|frames?|attempts?|retries?|requests?|users?|个|项|条|次|秒|分钟|小时|天|像素|字符|行|列|页|屏|帧|毫秒)"
|
|
933
|
+
r"(?![A-Za-z0-9_])",
|
|
934
|
+
re.I,
|
|
935
|
+
)
|
|
936
|
+
_ACTOR_MODIFIER_AFTER = re.compile(
|
|
937
|
+
r"^(?:[-‑](?:facing|visible|authored|generated|provided)\b"
|
|
938
|
+
r"|\s+(?:copy|language|interface|research)\b"
|
|
939
|
+
r"|\s+experience\s+research\b"
|
|
940
|
+
r"|(?:['’]s?|s['’])\s+needs\b"
|
|
941
|
+
r"|\s+needs\b(?!\s+to\b))",
|
|
942
|
+
re.I,
|
|
943
|
+
)
|
|
944
|
+
_CHINESE_USER_NON_ACTOR_AFTER = re.compile(
|
|
945
|
+
r"^(?:的?需求|文案|语言|界面|可见|知道|看到|研究)"
|
|
946
|
+
)
|
|
947
|
+
_BLOCK_MODAL_BEFORE = re.compile(
|
|
948
|
+
r"(?:\b(?:must|shall|should|may|can|will|would|to)\b(?:\s+[A-Za-z-]+){0,3}|(?:必须|应当|应该|可以|可)(?:\S{0,6}))\s*$",
|
|
949
|
+
re.I,
|
|
950
|
+
)
|
|
951
|
+
_FINAL_TERMINAL_NOUN = re.compile(
|
|
952
|
+
r"^\s+(?:approval|acceptance|artifact|checkpoint|response|verdict|submission|state|content|screen|result|release|merge|commit|push|ready|completion)\b",
|
|
953
|
+
re.I,
|
|
954
|
+
)
|
|
955
|
+
_CLOSED_FINAL_RELATION_SOURCE = re.compile(
|
|
956
|
+
r"^(?:the\s+)?final\s+skill\s+(?:must|shall)\s+(?:stay|remain)\s+generic[.!]?$",
|
|
957
|
+
re.I,
|
|
958
|
+
)
|
|
959
|
+
|
|
960
|
+
POLARITY_GROUPS: dict[str, list[tuple[re.Pattern[str], re.Pattern[str]]]] = {
|
|
961
|
+
"recency": [
|
|
962
|
+
(
|
|
963
|
+
re.compile(r"\b(?:before|prior)\b|之前|先", re.I),
|
|
964
|
+
re.compile(r"\b(?:after|subsequent)\b|之后|再", re.I),
|
|
965
|
+
),
|
|
966
|
+
(
|
|
967
|
+
re.compile(r"\b(?:latest|final)\b|最新|最终", re.I),
|
|
968
|
+
re.compile(r"\b(?:first|initial)\b|首次", re.I),
|
|
969
|
+
),
|
|
970
|
+
],
|
|
971
|
+
"threshold": [
|
|
972
|
+
(
|
|
973
|
+
re.compile(
|
|
974
|
+
r"\b(?:at least|minimum|more than)\b|\b(?:above|over)\b(?=\s+\d)|至少|最少|不少于|高于",
|
|
975
|
+
re.I,
|
|
976
|
+
),
|
|
977
|
+
re.compile(
|
|
978
|
+
r"\b(?:at most|maximum|less than)\b|\b(?:below|under)\b(?=\s+\d)|至多|最多|不超过|低于",
|
|
979
|
+
re.I,
|
|
980
|
+
),
|
|
981
|
+
)
|
|
982
|
+
],
|
|
983
|
+
"scope": [
|
|
984
|
+
(
|
|
985
|
+
re.compile(
|
|
986
|
+
r"\b(?:not\s+a\s+substitute|limited\s+to|restricted\s+to|every|each|all|only|solely|exclusively|none)\b|每个|每条|全部|所有|仅|无一",
|
|
987
|
+
re.I,
|
|
988
|
+
),
|
|
989
|
+
re.compile(r"\bany\b|任何", re.I),
|
|
990
|
+
)
|
|
991
|
+
],
|
|
992
|
+
}
|
|
993
|
+
|
|
994
|
+
|
|
995
|
+
def _dedupe_terms(matches: Iterable[str]) -> list[str]:
|
|
996
|
+
seen: set[str] = set()
|
|
997
|
+
values: list[str] = []
|
|
998
|
+
for value in matches:
|
|
999
|
+
if value in seen:
|
|
1000
|
+
continue
|
|
1001
|
+
seen.add(value)
|
|
1002
|
+
values.append(value)
|
|
1003
|
+
return values
|
|
1004
|
+
|
|
1005
|
+
|
|
1006
|
+
_QUOTED_LITERAL = re.compile(
|
|
1007
|
+
r'"(?:\\.|[^"\\])*"'
|
|
1008
|
+
r"|(?<![A-Za-z0-9])'(?:\\.|[^'\\])*'(?![A-Za-z0-9])"
|
|
1009
|
+
r"|“[^”]*”|‘[^’]*’"
|
|
1010
|
+
)
|
|
1011
|
+
_PATH_LITERAL = re.compile(
|
|
1012
|
+
r"(?<![A-Za-z0-9_./-])"
|
|
1013
|
+
r"(?:(?:\.{0,2}/|/)[^\s`\"']+|[^\s`\"']+/[^\s`\"']+\.[A-Za-z0-9]{1,8})"
|
|
1014
|
+
)
|
|
1015
|
+
|
|
1016
|
+
|
|
1017
|
+
def _mask_threshold_literals(text: str) -> str:
|
|
1018
|
+
"""Blank inline code, quoted literals, and path tokens at stable offsets."""
|
|
1019
|
+
chars = list(text)
|
|
1020
|
+
|
|
1021
|
+
cursor = 0
|
|
1022
|
+
while cursor < len(text):
|
|
1023
|
+
start = text.find("`", cursor)
|
|
1024
|
+
if start < 0:
|
|
1025
|
+
break
|
|
1026
|
+
run = 1
|
|
1027
|
+
while start + run < len(text) and text[start + run] == "`":
|
|
1028
|
+
run += 1
|
|
1029
|
+
marker = "`" * run
|
|
1030
|
+
end = text.find(marker, start + run)
|
|
1031
|
+
if end < 0:
|
|
1032
|
+
cursor = start + run
|
|
1033
|
+
continue
|
|
1034
|
+
_blank(chars, start, end + run)
|
|
1035
|
+
cursor = end + run
|
|
1036
|
+
|
|
1037
|
+
for pattern in (_QUOTED_LITERAL, _PATH_LITERAL):
|
|
1038
|
+
visible = "".join(chars)
|
|
1039
|
+
for match in pattern.finditer(visible):
|
|
1040
|
+
_blank(chars, match.start(), match.end())
|
|
1041
|
+
return "".join(chars)
|
|
1042
|
+
|
|
1043
|
+
|
|
1044
|
+
def _threshold_terms(text: str) -> list[str]:
|
|
1045
|
+
searchable = _mask_threshold_literals(text)
|
|
1046
|
+
matches: list[tuple[int, int, str]] = []
|
|
1047
|
+
occupied: list[tuple[int, int]] = []
|
|
1048
|
+
for pattern in (_THRESHOLD_DIMENSION, _THRESHOLD_RATIO, _THRESHOLD_QUANTITY):
|
|
1049
|
+
for match in pattern.finditer(searchable):
|
|
1050
|
+
if any(match.start() < end and start < match.end() for start, end in occupied):
|
|
1051
|
+
continue
|
|
1052
|
+
occupied.append((match.start(), match.end()))
|
|
1053
|
+
matches.append((match.start(), match.end(), match.group(0)))
|
|
1054
|
+
for pattern in (
|
|
1055
|
+
_THRESHOLD_COMPARATOR,
|
|
1056
|
+
_THRESHOLD_DIRECTIONAL_COMPARATOR,
|
|
1057
|
+
_THRESHOLD_WITHIN,
|
|
1058
|
+
):
|
|
1059
|
+
for match in pattern.finditer(searchable):
|
|
1060
|
+
matches.append((match.start(), match.end(), match.group(0)))
|
|
1061
|
+
matches.sort(key=lambda item: (item[0], item[1]))
|
|
1062
|
+
return _dedupe_terms(value for _, _, value in matches)
|
|
1063
|
+
|
|
1064
|
+
|
|
1065
|
+
def _recency_terms(text: str) -> list[str]:
|
|
1066
|
+
values = []
|
|
1067
|
+
for match in QUALIFIER_PATTERNS["recency"].finditer(text):
|
|
1068
|
+
if match.group(0).casefold() == "first" and re.match(
|
|
1069
|
+
r"[-‑]screen\b", text[match.end() :], re.I
|
|
1070
|
+
):
|
|
1071
|
+
continue
|
|
1072
|
+
values.append(match.group(0))
|
|
1073
|
+
return _dedupe_terms(values)
|
|
1074
|
+
|
|
1075
|
+
|
|
1076
|
+
def _scope_terms(text: str) -> list[str]:
|
|
1077
|
+
values = []
|
|
1078
|
+
for match in QUALIFIER_PATTERNS["scope"].finditer(text):
|
|
1079
|
+
term = match.group(0)
|
|
1080
|
+
before = text[: match.start()]
|
|
1081
|
+
if term.casefold() == "only" and re.search(
|
|
1082
|
+
r"\bnot\s+(?:[*_~]{1,3})?$", before, re.I
|
|
1083
|
+
):
|
|
1084
|
+
continue
|
|
1085
|
+
if term == "仅" and re.search(r"不(?:仅)?$", before):
|
|
1086
|
+
continue
|
|
1087
|
+
if term.casefold() == "per" and re.match(
|
|
1088
|
+
r"\s+(?:[*_~]{1,3})?the\b", text[match.end() :], re.I
|
|
1089
|
+
):
|
|
1090
|
+
# "per the classes/design above" is an according-to cross-reference,
|
|
1091
|
+
# not a closed-set quantifier such as "per affected client".
|
|
1092
|
+
continue
|
|
1093
|
+
values.append(term)
|
|
1094
|
+
return _dedupe_terms(values)
|
|
1095
|
+
|
|
1096
|
+
|
|
1097
|
+
def _actor_terms(text: str) -> list[str]:
|
|
1098
|
+
values = []
|
|
1099
|
+
for match in QUALIFIER_PATTERNS["actor"].finditer(text):
|
|
1100
|
+
term = match.group(0)
|
|
1101
|
+
if term.casefold() in {"user", "users"}:
|
|
1102
|
+
if _ACTOR_MODIFIER_AFTER.match(text[match.end() :]):
|
|
1103
|
+
continue
|
|
1104
|
+
elif term == "用户":
|
|
1105
|
+
if _CHINESE_USER_NON_ACTOR_AFTER.match(text[match.end() :]):
|
|
1106
|
+
continue
|
|
1107
|
+
elif term.casefold() == "agent" and re.match(
|
|
1108
|
+
r"[-‑](?:contract|facing|generated|owned)\b",
|
|
1109
|
+
text[match.end() :],
|
|
1110
|
+
re.I,
|
|
1111
|
+
):
|
|
1112
|
+
continue
|
|
1113
|
+
elif term.casefold() == "reviewer" and re.match(
|
|
1114
|
+
r"\s+(?:checks?|checklist|criteria|检查项)\b",
|
|
1115
|
+
text[match.end() :],
|
|
1116
|
+
re.I,
|
|
1117
|
+
):
|
|
1118
|
+
continue
|
|
1119
|
+
values.append(term)
|
|
1120
|
+
return _dedupe_terms(values)
|
|
1121
|
+
|
|
1122
|
+
|
|
1123
|
+
def _inside_code_path(text: str, start: int, end: int) -> bool:
|
|
1124
|
+
if text[:start].count("`") % 2 == 0:
|
|
1125
|
+
return False
|
|
1126
|
+
left = text.rfind("`", 0, start)
|
|
1127
|
+
right = text.find("`", end)
|
|
1128
|
+
if left < 0 or right < 0:
|
|
1129
|
+
return False
|
|
1130
|
+
token = text[left + 1 : right]
|
|
1131
|
+
return "/" in token or re.search(r"(?:^|[-_.])(?:md|txt|json|ya?ml|py|sh)$", token, re.I) is not None
|
|
1132
|
+
|
|
1133
|
+
|
|
1134
|
+
def _block_is_enumerated_noun(text: str, start: int, end: int) -> bool:
|
|
1135
|
+
before = text[max(0, start - 48) : start]
|
|
1136
|
+
after = text[end : end + 48]
|
|
1137
|
+
if _BLOCK_MODAL_BEFORE.search(before):
|
|
1138
|
+
return False
|
|
1139
|
+
previous = before.rstrip()[-1:] if before.rstrip() else ""
|
|
1140
|
+
following = after.lstrip()
|
|
1141
|
+
return previous in {",", "/"} and bool(
|
|
1142
|
+
re.match(r"^(?:,|/|\band\b|\bor\b|,|、|和|或)", following, re.I)
|
|
1143
|
+
)
|
|
1144
|
+
|
|
1145
|
+
|
|
1146
|
+
def _block_is_code_noun(text: str, start: int) -> bool:
|
|
1147
|
+
return re.search(r"\bcode\s+$", text[max(0, start - 24) : start], re.I) is not None
|
|
1148
|
+
|
|
1149
|
+
|
|
1150
|
+
def _red_is_color_term(text: str, start: int, end: int) -> bool:
|
|
1151
|
+
before = text[max(0, start - 32) : start]
|
|
1152
|
+
after = text[end : end + 32]
|
|
1153
|
+
return bool(
|
|
1154
|
+
re.search(r"\b(?:color|colour)(?:\s+is|\s*:)?\s*$", before, re.I)
|
|
1155
|
+
or re.match(r"^[-‑]colou?red\b", after, re.I)
|
|
1156
|
+
or re.match(r"^(?:[-‑]|\s+)(?:first|baseline)\b", after, re.I)
|
|
1157
|
+
or re.match(
|
|
1158
|
+
r"^\s+(?:color|colour|text|border|background|icon|badge|token|fill|stroke)\b",
|
|
1159
|
+
after,
|
|
1160
|
+
re.I,
|
|
1161
|
+
)
|
|
1162
|
+
)
|
|
1163
|
+
|
|
1164
|
+
|
|
1165
|
+
def _consequence_terms(text: str) -> list[str]:
|
|
1166
|
+
values = []
|
|
1167
|
+
for match in QUALIFIER_PATTERNS["consequence"].finditer(text):
|
|
1168
|
+
term = match.group(0)
|
|
1169
|
+
folded = term.casefold()
|
|
1170
|
+
if _inside_code_path(text, match.start(), match.end()):
|
|
1171
|
+
continue
|
|
1172
|
+
if folded == "block" and _block_is_enumerated_noun(
|
|
1173
|
+
text, match.start(), match.end()
|
|
1174
|
+
):
|
|
1175
|
+
continue
|
|
1176
|
+
if folded in {"block", "blocks"} and _block_is_code_noun(
|
|
1177
|
+
text, match.start()
|
|
1178
|
+
):
|
|
1179
|
+
continue
|
|
1180
|
+
if folded == "fail" and (
|
|
1181
|
+
text[match.start() - 1 : match.start()] == "/"
|
|
1182
|
+
or text[match.end() : match.end() + 1] == "/"
|
|
1183
|
+
):
|
|
1184
|
+
continue
|
|
1185
|
+
if folded == "red" and _red_is_color_term(
|
|
1186
|
+
text, match.start(), match.end()
|
|
1187
|
+
):
|
|
1188
|
+
continue
|
|
1189
|
+
values.append(term)
|
|
1190
|
+
return _dedupe_terms(values)
|
|
1191
|
+
|
|
1192
|
+
|
|
1193
|
+
def qualifier_terms(text: str) -> dict[str, list[str]]:
|
|
1194
|
+
extractors = {
|
|
1195
|
+
"modality": lambda value: _dedupe_terms(
|
|
1196
|
+
match.group(0) for match in QUALIFIER_PATTERNS["modality"].finditer(value)
|
|
1197
|
+
),
|
|
1198
|
+
"recency": _recency_terms,
|
|
1199
|
+
"threshold": _threshold_terms,
|
|
1200
|
+
"scope": _scope_terms,
|
|
1201
|
+
"actor": _actor_terms,
|
|
1202
|
+
"consequence": _consequence_terms,
|
|
1203
|
+
}
|
|
1204
|
+
result: dict[str, list[str]] = {}
|
|
1205
|
+
for kind in ("modality", "recency", "threshold", "scope", "actor", "consequence"):
|
|
1206
|
+
terms = extractors[kind](text)
|
|
1207
|
+
if terms:
|
|
1208
|
+
result[kind] = terms
|
|
1209
|
+
return result
|
|
1210
|
+
|
|
1211
|
+
|
|
1212
|
+
_QUALIFIER_RELATION_FIELDS = {
|
|
1213
|
+
"kind",
|
|
1214
|
+
"term",
|
|
1215
|
+
"occurrence",
|
|
1216
|
+
"source_excerpt",
|
|
1217
|
+
"resolution",
|
|
1218
|
+
"same_immediate_host",
|
|
1219
|
+
}
|
|
1220
|
+
_DIRECTIONAL_RECENCY_TERMS = {
|
|
1221
|
+
"before",
|
|
1222
|
+
"after",
|
|
1223
|
+
"prior",
|
|
1224
|
+
"subsequent",
|
|
1225
|
+
"之前",
|
|
1226
|
+
"之后",
|
|
1227
|
+
"先",
|
|
1228
|
+
"再",
|
|
1229
|
+
}
|
|
1230
|
+
|
|
1231
|
+
|
|
1232
|
+
def _literal_occurrences(text: str, term: str) -> list[re.Match[str]]:
|
|
1233
|
+
return list(
|
|
1234
|
+
re.finditer(
|
|
1235
|
+
rf"(?<![A-Za-z0-9_]){re.escape(term)}(?![A-Za-z0-9_])",
|
|
1236
|
+
text,
|
|
1237
|
+
re.I,
|
|
1238
|
+
)
|
|
1239
|
+
)
|
|
1240
|
+
|
|
1241
|
+
|
|
1242
|
+
def _is_closed_non_temporal_final_relation(
|
|
1243
|
+
source_text: str, carrier_text: str, occurrence: re.Match[str]
|
|
1244
|
+
) -> bool:
|
|
1245
|
+
normalized_source = GCD.normalize(source_text)
|
|
1246
|
+
if not _CLOSED_FINAL_RELATION_SOURCE.fullmatch(normalized_source):
|
|
1247
|
+
return False
|
|
1248
|
+
rewritten = (
|
|
1249
|
+
normalized_source[: occurrence.start()]
|
|
1250
|
+
+ "resulting"
|
|
1251
|
+
+ normalized_source[occurrence.end() :]
|
|
1252
|
+
)
|
|
1253
|
+
return GCD.normalize(rewritten).casefold() == GCD.normalize(carrier_text).casefold()
|
|
1254
|
+
|
|
1255
|
+
|
|
1256
|
+
def validate_qualifier_relations(
|
|
1257
|
+
expected: ExpectedRow,
|
|
1258
|
+
mapping: dict[str, Any],
|
|
1259
|
+
carrier: Carrier,
|
|
1260
|
+
required: dict[str, list[str]],
|
|
1261
|
+
) -> dict[str, list[str]]:
|
|
1262
|
+
label = f"{expected.source_path}#{expected.source_ordinal}"
|
|
1263
|
+
relations = mapping.get("qualifier_relations", [])
|
|
1264
|
+
if not isinstance(relations, list):
|
|
1265
|
+
raise AuditError("QUALIFIER_RELATION_INVALID", f"{label}: array required")
|
|
1266
|
+
if not relations:
|
|
1267
|
+
return required
|
|
1268
|
+
if mapping.get("effect") != "strengthened":
|
|
1269
|
+
raise AuditError("QUALIFIER_RELATION_REQUIRES_STRENGTHENED", label)
|
|
1270
|
+
if mapping.get("manual_reviewed") is not True:
|
|
1271
|
+
raise AuditError("STRENGTHENED_REVIEW_REQUIRED", label)
|
|
1272
|
+
|
|
1273
|
+
effective = {kind: list(terms) for kind, terms in required.items()}
|
|
1274
|
+
seen: set[tuple[str, str, int]] = set()
|
|
1275
|
+
for relation in relations:
|
|
1276
|
+
if not isinstance(relation, dict) or set(relation) != _QUALIFIER_RELATION_FIELDS:
|
|
1277
|
+
raise AuditError("QUALIFIER_RELATION_INVALID", label)
|
|
1278
|
+
kind = relation.get("kind")
|
|
1279
|
+
if kind != "recency":
|
|
1280
|
+
raise AuditError(
|
|
1281
|
+
"QUALIFIER_RELATION_FORBIDDEN", f"{label}: kind={kind}"
|
|
1282
|
+
)
|
|
1283
|
+
if relation.get("resolution") != "carrier-applies-unconditionally":
|
|
1284
|
+
raise AuditError("QUALIFIER_RELATION_INVALID", f"{label}: resolution")
|
|
1285
|
+
if relation.get("same_immediate_host") is not True:
|
|
1286
|
+
raise AuditError("QUALIFIER_RELATION_INVALID", f"{label}: host")
|
|
1287
|
+
term = relation.get("term")
|
|
1288
|
+
occurrence = relation.get("occurrence")
|
|
1289
|
+
excerpt = relation.get("source_excerpt")
|
|
1290
|
+
if (
|
|
1291
|
+
not isinstance(term, str)
|
|
1292
|
+
or not term
|
|
1293
|
+
or not isinstance(occurrence, int)
|
|
1294
|
+
or isinstance(occurrence, bool)
|
|
1295
|
+
or occurrence < 1
|
|
1296
|
+
or not isinstance(excerpt, str)
|
|
1297
|
+
or not excerpt
|
|
1298
|
+
or "\n" in excerpt
|
|
1299
|
+
or len(excerpt) > 120
|
|
1300
|
+
):
|
|
1301
|
+
raise AuditError("QUALIFIER_RELATION_INVALID", label)
|
|
1302
|
+
identity = (str(kind), term.casefold(), occurrence)
|
|
1303
|
+
if identity in seen:
|
|
1304
|
+
raise AuditError("QUALIFIER_RELATION_INVALID", f"{label}: duplicate")
|
|
1305
|
+
seen.add(identity)
|
|
1306
|
+
|
|
1307
|
+
source_terms = effective.get("recency", [])
|
|
1308
|
+
matching_source_terms = [
|
|
1309
|
+
value for value in source_terms if value.casefold() == term.casefold()
|
|
1310
|
+
]
|
|
1311
|
+
occurrences = _literal_occurrences(expected.before_text, term)
|
|
1312
|
+
excerpt_start = expected.before_text.find(excerpt)
|
|
1313
|
+
if (
|
|
1314
|
+
len(matching_source_terms) != 1
|
|
1315
|
+
or len(occurrences) != 1
|
|
1316
|
+
or occurrence != 1
|
|
1317
|
+
or expected.before_text.count(excerpt) != 1
|
|
1318
|
+
or excerpt_start < 0
|
|
1319
|
+
or not (
|
|
1320
|
+
excerpt_start <= occurrences[0].start()
|
|
1321
|
+
and occurrences[0].end() <= excerpt_start + len(excerpt)
|
|
1322
|
+
)
|
|
1323
|
+
):
|
|
1324
|
+
raise AuditError("QUALIFIER_RELATION_INVALID", label)
|
|
1325
|
+
|
|
1326
|
+
other_recency = [
|
|
1327
|
+
value for value in source_terms if value.casefold() != term.casefold()
|
|
1328
|
+
]
|
|
1329
|
+
source_consequences = required.get("consequence", [])
|
|
1330
|
+
after_recency = qualifier_terms(carrier.text).get("recency", [])
|
|
1331
|
+
terminal_tail = expected.before_text[occurrences[0].end() :]
|
|
1332
|
+
if (
|
|
1333
|
+
term.casefold() != "final"
|
|
1334
|
+
or not _is_closed_non_temporal_final_relation(
|
|
1335
|
+
expected.before_text, carrier.text, occurrences[0]
|
|
1336
|
+
)
|
|
1337
|
+
or term.casefold() in _DIRECTIONAL_RECENCY_TERMS
|
|
1338
|
+
or other_recency
|
|
1339
|
+
or source_consequences
|
|
1340
|
+
or _FINAL_TERMINAL_NOUN.match(terminal_tail)
|
|
1341
|
+
or after_recency
|
|
1342
|
+
):
|
|
1343
|
+
raise AuditError("QUALIFIER_RELATION_FORBIDDEN", label)
|
|
1344
|
+
|
|
1345
|
+
effective["recency"] = [
|
|
1346
|
+
value for value in source_terms if value.casefold() != term.casefold()
|
|
1347
|
+
]
|
|
1348
|
+
if not effective["recency"]:
|
|
1349
|
+
del effective["recency"]
|
|
1350
|
+
return effective
|
|
1351
|
+
|
|
1352
|
+
|
|
1353
|
+
_CLAUSE_ACTION = re.compile(
|
|
1354
|
+
r"\b(?:must|shall|required|record|verify|preserve|reject|block|expose|show|keep|route|check|name|include|cover|provide|ensure|prevent|allow|forbid)\b|必须|应当|记录|验证|保留|拒绝|阻塞|展示|路由|检查|命名|包括|覆盖|提供|确保|防止|允许|禁止",
|
|
1355
|
+
re.I,
|
|
1356
|
+
)
|
|
1357
|
+
_CLAUSE_LOCAL_ACTION = re.compile(
|
|
1358
|
+
r"\b(?:record|verify|preserve|reject|block|expose|show|keep|route|check|name|include|cover|provide|ensure|prevent|allow|forbid|retain|confirm|complete|finish|remain|stay)\b|记录|验证|保留|拒绝|阻塞|展示|路由|检查|命名|包括|覆盖|提供|确保|防止|允许|禁止|保有|确认|完成|保持|测试",
|
|
1359
|
+
re.I,
|
|
1360
|
+
)
|
|
1361
|
+
|
|
1362
|
+
|
|
1363
|
+
def _mask_code_spans_text(text: str) -> str:
|
|
1364
|
+
"""Same-length copy with inline code-span characters replaced by NUL.
|
|
1365
|
+
|
|
1366
|
+
Clause delimiters (semicolons, colons, commas, join words) inside inline
|
|
1367
|
+
code are literal content, not clause structure, so structural scans run
|
|
1368
|
+
on this masked copy while clause text is sliced from the original.
|
|
1369
|
+
"""
|
|
1370
|
+
mask = code_span_mask(text)
|
|
1371
|
+
return "".join(
|
|
1372
|
+
"\x00" if in_code else char for char, in_code in zip(text, mask)
|
|
1373
|
+
)
|
|
1374
|
+
|
|
1375
|
+
|
|
1376
|
+
def compound_clauses(text: str) -> list[dict[str, str]]:
|
|
1377
|
+
"""Conservatively and reproducibly split a baseline compound obligation."""
|
|
1378
|
+
normalized = GCD.normalize(text)
|
|
1379
|
+
structural = _mask_code_spans_text(normalized)
|
|
1380
|
+
|
|
1381
|
+
def split_outside_code(value: str, value_structural: str, pattern: str) -> list[str]:
|
|
1382
|
+
parts: list[str] = []
|
|
1383
|
+
start = 0
|
|
1384
|
+
for match in re.finditer(pattern, value_structural):
|
|
1385
|
+
parts.append(value[start : match.start()])
|
|
1386
|
+
start = match.end()
|
|
1387
|
+
parts.append(value[start:])
|
|
1388
|
+
return parts
|
|
1389
|
+
|
|
1390
|
+
semicolon_parts = [
|
|
1391
|
+
GCD.normalize(part)
|
|
1392
|
+
for part in split_outside_code(normalized, structural, r"[;;]")
|
|
1393
|
+
]
|
|
1394
|
+
semicolon_parts = [part for part in semicolon_parts if part]
|
|
1395
|
+
if len(semicolon_parts) >= 2:
|
|
1396
|
+
return [
|
|
1397
|
+
{"id": f"c{index}", "text": part}
|
|
1398
|
+
for index, part in enumerate(semicolon_parts, 1)
|
|
1399
|
+
]
|
|
1400
|
+
|
|
1401
|
+
colon = re.match(r"^(.*?[::])\s*(.+)$", structural)
|
|
1402
|
+
if not colon or not _CLAUSE_ACTION.search(structural[: colon.end(1)]):
|
|
1403
|
+
return []
|
|
1404
|
+
tail = normalized[colon.start(2) :]
|
|
1405
|
+
tail_structural = structural[colon.start(2) :]
|
|
1406
|
+
final_join = re.search(
|
|
1407
|
+
r"(?:,|,)\s*(?:and|or|以及|与|或)\s+", tail_structural, re.I
|
|
1408
|
+
)
|
|
1409
|
+
if not final_join:
|
|
1410
|
+
return []
|
|
1411
|
+
expanded = tail[: final_join.start()] + ", " + tail[final_join.end() :]
|
|
1412
|
+
expanded_structural = (
|
|
1413
|
+
tail_structural[: final_join.start()]
|
|
1414
|
+
+ ", "
|
|
1415
|
+
+ tail_structural[final_join.end() :]
|
|
1416
|
+
)
|
|
1417
|
+
item_pairs = [
|
|
1418
|
+
(GCD.normalize(item), item_structural)
|
|
1419
|
+
for item, item_structural in zip(
|
|
1420
|
+
split_outside_code(expanded, expanded_structural, r"[,,]"),
|
|
1421
|
+
split_outside_code(expanded_structural, expanded_structural, r"[,,]"),
|
|
1422
|
+
)
|
|
1423
|
+
]
|
|
1424
|
+
item_pairs = [pair for pair in item_pairs if pair[0]]
|
|
1425
|
+
items = [item for item, _ in item_pairs]
|
|
1426
|
+
if len(items) < 3:
|
|
1427
|
+
return []
|
|
1428
|
+
clause_like = sum(
|
|
1429
|
+
bool(_CLAUSE_ACTION.search(item_structural))
|
|
1430
|
+
or GCD.effective_len(item) >= 24
|
|
1431
|
+
for item, item_structural in item_pairs
|
|
1432
|
+
)
|
|
1433
|
+
if clause_like < 2:
|
|
1434
|
+
return []
|
|
1435
|
+
parts = [normalized[: colon.end(1)] + " " + items[0], *items[1:]]
|
|
1436
|
+
return [
|
|
1437
|
+
{"id": f"c{index}", "text": part}
|
|
1438
|
+
for index, part in enumerate(parts, 1)
|
|
1439
|
+
]
|
|
1440
|
+
|
|
1441
|
+
|
|
1442
|
+
def ensure_qualifier_strength(before: str, after: str, label: str) -> None:
|
|
1443
|
+
required = qualifier_terms(before)
|
|
1444
|
+
available = qualifier_terms(after)
|
|
1445
|
+
missing = sorted(set(required) - set(available))
|
|
1446
|
+
if missing:
|
|
1447
|
+
raise AuditError(
|
|
1448
|
+
"BUNDLE_QUALIFIER_MISSING", f"{label}: classes={','.join(missing)}"
|
|
1449
|
+
)
|
|
1450
|
+
if HARD_MODALITY.search(before) and not HARD_MODALITY.search(after):
|
|
1451
|
+
raise AuditError("BUNDLE_QUALIFIER_WEAKENED", label)
|
|
1452
|
+
for polarity_kind in required:
|
|
1453
|
+
for positive, negative in POLARITY_GROUPS.get(polarity_kind, []):
|
|
1454
|
+
if positive.search(before) and negative.search(after):
|
|
1455
|
+
raise AuditError("BUNDLE_QUALIFIER_REVERSED", f"{label}: {polarity_kind}")
|
|
1456
|
+
if negative.search(before) and positive.search(after):
|
|
1457
|
+
raise AuditError("BUNDLE_QUALIFIER_REVERSED", f"{label}: {polarity_kind}")
|
|
1458
|
+
for kind, before_terms in required.items():
|
|
1459
|
+
before_literals = {term.casefold() for term in before_terms}
|
|
1460
|
+
after_literals = {term.casefold() for term in available.get(kind, [])}
|
|
1461
|
+
absent = sorted(before_literals - after_literals)
|
|
1462
|
+
if absent:
|
|
1463
|
+
raise AuditError(
|
|
1464
|
+
"BUNDLE_QUALIFIER_LITERAL_MISSING",
|
|
1465
|
+
f"{label}: {kind}={','.join(absent)}",
|
|
1466
|
+
)
|
|
1467
|
+
for kind in required:
|
|
1468
|
+
for positive, negative in POLARITY_GROUPS.get(kind, []):
|
|
1469
|
+
if positive.search(before) and negative.search(after):
|
|
1470
|
+
raise AuditError("BUNDLE_QUALIFIER_REVERSED", f"{label}: {kind}")
|
|
1471
|
+
if negative.search(before) and positive.search(after):
|
|
1472
|
+
raise AuditError("BUNDLE_QUALIFIER_REVERSED", f"{label}: {kind}")
|
|
1473
|
+
|
|
1474
|
+
|
|
1475
|
+
def _clause_action_terms(text: str) -> list[str]:
|
|
1476
|
+
return _dedupe_terms(
|
|
1477
|
+
match.group(0) for match in _CLAUSE_LOCAL_ACTION.finditer(text)
|
|
1478
|
+
)
|
|
1479
|
+
|
|
1480
|
+
|
|
1481
|
+
def _missing_terms(required: Iterable[str], available: Iterable[str]) -> list[str]:
|
|
1482
|
+
wanted = Counter(value.casefold() for value in required)
|
|
1483
|
+
present = Counter(value.casefold() for value in available)
|
|
1484
|
+
return sorted((wanted - present).elements())
|
|
1485
|
+
|
|
1486
|
+
|
|
1487
|
+
def validate_semantic_review(
|
|
1488
|
+
evidence: dict[str, Any], label: str, *, required: bool
|
|
1489
|
+
) -> None:
|
|
1490
|
+
"""Validate review-evidence shape, never the truth of the review decision."""
|
|
1491
|
+
status = evidence.get("semantic_review")
|
|
1492
|
+
rationale = evidence.get("semantic_rationale")
|
|
1493
|
+
if required:
|
|
1494
|
+
if status != "reviewed":
|
|
1495
|
+
raise AuditError("SEMANTIC_REVIEW_REQUIRED", label)
|
|
1496
|
+
if not isinstance(rationale, str) or not rationale.strip():
|
|
1497
|
+
raise AuditError("SEMANTIC_RATIONALE_REQUIRED", label)
|
|
1498
|
+
return
|
|
1499
|
+
if status is not None or rationale is not None:
|
|
1500
|
+
raise AuditError("SEMANTIC_REVIEW_REDUNDANT", label)
|
|
1501
|
+
|
|
1502
|
+
|
|
1503
|
+
def ensure_clause_local_lexical_guard(before: str, after: str, label: str) -> None:
|
|
1504
|
+
"""Catch literal local omissions without claiming semantic equivalence."""
|
|
1505
|
+
before_qualifiers = qualifier_terms(before)
|
|
1506
|
+
after_qualifiers = qualifier_terms(after)
|
|
1507
|
+
for kind in ("modality", "actor", "consequence"):
|
|
1508
|
+
missing = _missing_terms(
|
|
1509
|
+
before_qualifiers.get(kind, []), after_qualifiers.get(kind, [])
|
|
1510
|
+
)
|
|
1511
|
+
if missing:
|
|
1512
|
+
raise AuditError(
|
|
1513
|
+
f"BUNDLE_CLAUSE_{kind.upper()}_MISSING",
|
|
1514
|
+
f"{label}: {','.join(missing)}",
|
|
1515
|
+
)
|
|
1516
|
+
missing_actions = _missing_terms(
|
|
1517
|
+
_clause_action_terms(before), _clause_action_terms(after)
|
|
1518
|
+
)
|
|
1519
|
+
if missing_actions:
|
|
1520
|
+
raise AuditError(
|
|
1521
|
+
"BUNDLE_CLAUSE_ACTION_MISSING",
|
|
1522
|
+
f"{label}: {','.join(missing_actions)}",
|
|
1523
|
+
)
|
|
1524
|
+
|
|
1525
|
+
|
|
1526
|
+
def validate_bundle_clause_proof(
|
|
1527
|
+
before: str,
|
|
1528
|
+
after: str,
|
|
1529
|
+
claimed: Any,
|
|
1530
|
+
label: str,
|
|
1531
|
+
) -> None:
|
|
1532
|
+
# Open-text actor/action/consequence equivalence is review-owned. These
|
|
1533
|
+
# deterministic checks are lexical tripwires, not semantic proof.
|
|
1534
|
+
ensure_clause_local_lexical_guard(before, after, label)
|
|
1535
|
+
if not isinstance(claimed, list):
|
|
1536
|
+
raise AuditError("BUNDLE_CLAUSE_PROOF_INVALID", label)
|
|
1537
|
+
required = qualifier_terms(before)
|
|
1538
|
+
available = qualifier_terms(after)
|
|
1539
|
+
claimed_kinds: list[str] = []
|
|
1540
|
+
for item in claimed:
|
|
1541
|
+
if not isinstance(item, dict) or set(item) != {
|
|
1542
|
+
"kind",
|
|
1543
|
+
"before",
|
|
1544
|
+
"after",
|
|
1545
|
+
"same_immediate_host",
|
|
1546
|
+
}:
|
|
1547
|
+
raise AuditError("BUNDLE_CLAUSE_PROOF_INVALID", label)
|
|
1548
|
+
kind = item.get("kind")
|
|
1549
|
+
if kind not in QUALIFIER_KINDS or item.get("same_immediate_host") is not True:
|
|
1550
|
+
raise AuditError("BUNDLE_CLAUSE_PROOF_INVALID", f"{label}: {kind}")
|
|
1551
|
+
claimed_kinds.append(str(kind))
|
|
1552
|
+
before_values = item.get("before")
|
|
1553
|
+
after_values = item.get("after")
|
|
1554
|
+
if not isinstance(before_values, list) or not isinstance(after_values, list):
|
|
1555
|
+
raise AuditError("BUNDLE_CLAUSE_PROOF_INVALID", f"{label}: {kind}")
|
|
1556
|
+
if Counter(str(value).casefold() for value in before_values) != Counter(
|
|
1557
|
+
value.casefold() for value in required.get(str(kind), [])
|
|
1558
|
+
):
|
|
1559
|
+
raise AuditError("BUNDLE_QUALIFIER_BEFORE_INCOMPLETE", f"{label}: {kind}")
|
|
1560
|
+
if Counter(str(value).casefold() for value in after_values) != Counter(
|
|
1561
|
+
value.casefold() for value in available.get(str(kind), [])
|
|
1562
|
+
):
|
|
1563
|
+
raise AuditError("BUNDLE_QUALIFIER_AFTER_INCOMPLETE", f"{label}: {kind}")
|
|
1564
|
+
if len(claimed_kinds) != len(set(claimed_kinds)) or set(claimed_kinds) != set(required):
|
|
1565
|
+
raise AuditError("BUNDLE_QUALIFIER_CLASS_MISMATCH", label)
|
|
1566
|
+
ensure_qualifier_strength(before, after, label)
|
|
1567
|
+
|
|
1568
|
+
|
|
1569
|
+
def _aggregate_qualifier_terms(texts: Iterable[str]) -> dict[str, list[str]]:
|
|
1570
|
+
aggregate: dict[str, list[str]] = {}
|
|
1571
|
+
for text in texts:
|
|
1572
|
+
for kind, terms in qualifier_terms(text).items():
|
|
1573
|
+
aggregate.setdefault(kind, []).extend(terms)
|
|
1574
|
+
return {kind: _dedupe_terms(terms) for kind, terms in aggregate.items()}
|
|
1575
|
+
|
|
1576
|
+
|
|
1577
|
+
def validate_bundle_summary_proof(
|
|
1578
|
+
clause_pairs: Iterable[tuple[str, str]], claimed: Any, label: str
|
|
1579
|
+
) -> None:
|
|
1580
|
+
pairs = list(clause_pairs)
|
|
1581
|
+
required = _aggregate_qualifier_terms(before for before, _ in pairs)
|
|
1582
|
+
available = _aggregate_qualifier_terms(after for _, after in pairs)
|
|
1583
|
+
if not isinstance(claimed, list):
|
|
1584
|
+
raise AuditError("QUALIFIERS_INVALID", f"{label}: array required")
|
|
1585
|
+
claimed_kinds: list[str] = []
|
|
1586
|
+
for item in claimed:
|
|
1587
|
+
if not isinstance(item, dict) or item.get("kind") not in QUALIFIER_KINDS:
|
|
1588
|
+
raise AuditError("QUALIFIERS_INVALID", f"{label}: bad qualifier entry")
|
|
1589
|
+
kind = str(item["kind"])
|
|
1590
|
+
claimed_kinds.append(kind)
|
|
1591
|
+
if item.get("same_immediate_host") is not True:
|
|
1592
|
+
raise AuditError("QUALIFIER_WRONG_HOST", f"{label}: {kind}")
|
|
1593
|
+
before_values = item.get("before")
|
|
1594
|
+
after_values = item.get("after")
|
|
1595
|
+
if not isinstance(before_values, list) or not isinstance(after_values, list):
|
|
1596
|
+
raise AuditError("QUALIFIERS_INVALID", f"{label}: {kind}")
|
|
1597
|
+
if Counter(str(value).casefold() for value in before_values) != Counter(
|
|
1598
|
+
value.casefold() for value in required.get(kind, [])
|
|
1599
|
+
):
|
|
1600
|
+
raise AuditError("BUNDLE_QUALIFIER_BEFORE_INCOMPLETE", f"{label}: {kind}")
|
|
1601
|
+
if Counter(str(value).casefold() for value in after_values) != Counter(
|
|
1602
|
+
value.casefold() for value in available.get(kind, [])
|
|
1603
|
+
):
|
|
1604
|
+
raise AuditError("BUNDLE_QUALIFIER_AFTER_INCOMPLETE", f"{label}: {kind}")
|
|
1605
|
+
if len(claimed_kinds) != len(set(claimed_kinds)) or set(claimed_kinds) != set(
|
|
1606
|
+
required
|
|
1607
|
+
):
|
|
1608
|
+
raise AuditError("BUNDLE_QUALIFIER_CLASS_MISMATCH", label)
|
|
1609
|
+
|
|
1610
|
+
|
|
1611
|
+
def validate_qualifiers(
|
|
1612
|
+
expected: ExpectedRow, mapping: dict[str, Any], carrier: Carrier
|
|
1613
|
+
) -> None:
|
|
1614
|
+
label = f"{expected.source_path}#{expected.source_ordinal}"
|
|
1615
|
+
before = expected.before_text
|
|
1616
|
+
after = carrier.text
|
|
1617
|
+
required = validate_qualifier_relations(
|
|
1618
|
+
expected, mapping, carrier, qualifier_terms(before)
|
|
1619
|
+
)
|
|
1620
|
+
claimed = mapping.get("qualifiers")
|
|
1621
|
+
if not isinstance(claimed, list):
|
|
1622
|
+
raise AuditError("QUALIFIERS_INVALID", f"{label}: array required")
|
|
1623
|
+
|
|
1624
|
+
# A byte-for-byte normative carrier proves its own qualifier preservation.
|
|
1625
|
+
if GCD.normalize(before) == GCD.normalize(after):
|
|
1626
|
+
if claimed:
|
|
1627
|
+
raise AuditError("QUALIFIERS_REDUNDANT", f"{label}: verbatim carrier")
|
|
1628
|
+
return
|
|
1629
|
+
|
|
1630
|
+
if mapping.get("manual_reviewed") is not True:
|
|
1631
|
+
raise AuditError("MANUAL_REVIEW_REQUIRED", f"{label}: rephrased carrier")
|
|
1632
|
+
claimed_kinds: list[str] = []
|
|
1633
|
+
for item in claimed:
|
|
1634
|
+
if not isinstance(item, dict) or item.get("kind") not in QUALIFIER_KINDS:
|
|
1635
|
+
raise AuditError("QUALIFIERS_INVALID", f"{label}: bad qualifier entry")
|
|
1636
|
+
kind = str(item["kind"])
|
|
1637
|
+
claimed_kinds.append(kind)
|
|
1638
|
+
if item.get("same_immediate_host") is not True:
|
|
1639
|
+
raise AuditError("QUALIFIER_WRONG_HOST", f"{label}: {kind}")
|
|
1640
|
+
before_terms = item.get("before")
|
|
1641
|
+
after_terms = item.get("after")
|
|
1642
|
+
if not isinstance(before_terms, list) or not before_terms:
|
|
1643
|
+
raise AuditError("QUALIFIERS_INVALID", f"{label}: {kind}.before")
|
|
1644
|
+
if not isinstance(after_terms, list) or not after_terms:
|
|
1645
|
+
raise AuditError("QUALIFIERS_INVALID", f"{label}: {kind}.after")
|
|
1646
|
+
wanted_before = Counter(term.casefold() for term in required.get(kind, []))
|
|
1647
|
+
claimed_before = Counter(GCD.normalize(str(term)).casefold() for term in before_terms)
|
|
1648
|
+
if claimed_before != wanted_before:
|
|
1649
|
+
raise AuditError(
|
|
1650
|
+
"QUALIFIER_BEFORE_INCOMPLETE",
|
|
1651
|
+
f"{label}: {kind} expected {list(wanted_before.elements())}, got {list(claimed_before.elements())}",
|
|
1652
|
+
)
|
|
1653
|
+
extracted_after = qualifier_terms(after).get(kind, [])
|
|
1654
|
+
wanted_after = Counter(term.casefold() for term in extracted_after)
|
|
1655
|
+
claimed_after = Counter(GCD.normalize(str(term)).casefold() for term in after_terms)
|
|
1656
|
+
if claimed_after != wanted_after:
|
|
1657
|
+
raise AuditError(
|
|
1658
|
+
"QUALIFIER_AFTER_INCOMPLETE",
|
|
1659
|
+
f"{label}: {kind} expected {list(wanted_after.elements())}, got {list(claimed_after.elements())}",
|
|
1660
|
+
)
|
|
1661
|
+
for positive, negative in POLARITY_GROUPS.get(kind, []):
|
|
1662
|
+
if positive.search(before) and negative.search(after):
|
|
1663
|
+
raise AuditError("QUALIFIER_REVERSED", f"{label}: {kind}")
|
|
1664
|
+
if negative.search(before) and positive.search(after):
|
|
1665
|
+
raise AuditError("QUALIFIER_REVERSED", f"{label}: {kind}")
|
|
1666
|
+
if len(claimed_kinds) != len(set(claimed_kinds)) or set(claimed_kinds) != set(required):
|
|
1667
|
+
raise AuditError(
|
|
1668
|
+
"QUALIFIER_CLASS_MISMATCH",
|
|
1669
|
+
f"{label}: expected {sorted(required)}, got {sorted(claimed_kinds)}",
|
|
1670
|
+
)
|
|
1671
|
+
if HARD_MODALITY.search(before) and not HARD_MODALITY.search(after):
|
|
1672
|
+
if SOFT_MODALITY.search(after):
|
|
1673
|
+
raise AuditError("QUALIFIER_WEAKENED", f"{label}: hard modality became soft")
|
|
1674
|
+
raise AuditError("QUALIFIER_DROPPED", f"{label}: hard modality absent")
|
|
1675
|
+
|
|
1676
|
+
|
|
1677
|
+
BundleCarrier = list[tuple[Carrier, tuple[str, ...]]]
|
|
1678
|
+
PartitionedCarrier = list[PartitionedPart]
|
|
1679
|
+
|
|
1680
|
+
|
|
1681
|
+
def carrier_digest(path: str, chain: Iterable[str], text: str) -> str:
|
|
1682
|
+
payload = json.dumps(
|
|
1683
|
+
[normalize_path(path), list(chain), GCD.normalize(text)],
|
|
1684
|
+
ensure_ascii=False,
|
|
1685
|
+
separators=(",", ":"),
|
|
1686
|
+
).encode("utf-8")
|
|
1687
|
+
return hashlib.sha256(payload).hexdigest()
|
|
1688
|
+
|
|
1689
|
+
|
|
1690
|
+
def _span_boundary_ok(text: str, offset: int) -> bool:
|
|
1691
|
+
if offset <= 0 or offset >= len(text):
|
|
1692
|
+
return True
|
|
1693
|
+
return not (
|
|
1694
|
+
re.fullmatch(r"[A-Za-z0-9_]", text[offset - 1])
|
|
1695
|
+
and re.fullmatch(r"[A-Za-z0-9_]", text[offset])
|
|
1696
|
+
)
|
|
1697
|
+
|
|
1698
|
+
|
|
1699
|
+
def _literal_bridge_present(source_text: str, carrier_text: str, term: str) -> bool:
|
|
1700
|
+
normalized = GCD.normalize(term)
|
|
1701
|
+
if not normalized or len(normalized) < 2:
|
|
1702
|
+
return False
|
|
1703
|
+
pattern = re.compile(
|
|
1704
|
+
rf"(?<![A-Za-z0-9_]){re.escape(normalized)}(?![A-Za-z0-9_])",
|
|
1705
|
+
re.I,
|
|
1706
|
+
)
|
|
1707
|
+
return bool(pattern.search(source_text) and pattern.search(carrier_text))
|
|
1708
|
+
|
|
1709
|
+
|
|
1710
|
+
_PART_QUALIFIER_RESOLUTION_FIELDS = {"kind", "before", "resolution"}
|
|
1711
|
+
_IMPERATIVE_NORMATIVE = re.compile(
|
|
1712
|
+
r"^(?:apply|build|choose|classify|define|distinguish|expose|keep|load|map|name|"
|
|
1713
|
+
r"prevent|prefer|record|reserve|show|split|state|treat|use|verify)\b",
|
|
1714
|
+
re.I,
|
|
1715
|
+
)
|
|
1716
|
+
|
|
1717
|
+
|
|
1718
|
+
def _implicit_normative_strength(
|
|
1719
|
+
source_term: str, carriers: Iterable[Carrier]
|
|
1720
|
+
) -> bool:
|
|
1721
|
+
hard = bool(HARD_MODALITY.fullmatch(source_term)) or source_term in {
|
|
1722
|
+
"必须",
|
|
1723
|
+
"不得",
|
|
1724
|
+
"禁止",
|
|
1725
|
+
"不可",
|
|
1726
|
+
"不能",
|
|
1727
|
+
}
|
|
1728
|
+
for carrier in carriers:
|
|
1729
|
+
text = GCD.normalize(carrier.text)
|
|
1730
|
+
if SOFT_MODALITY.search(text):
|
|
1731
|
+
continue
|
|
1732
|
+
if HARD_MODALITY.search(text):
|
|
1733
|
+
return True
|
|
1734
|
+
if _IMPERATIVE_NORMATIVE.match(text) or re.match(r"^(?:Every|Each)\b", text):
|
|
1735
|
+
return True
|
|
1736
|
+
chain = " > ".join(carrier.chain)
|
|
1737
|
+
if not hard and re.search(
|
|
1738
|
+
r"(?:Hard design rules|Design quality checks|Acceptance|operative criteria)",
|
|
1739
|
+
chain,
|
|
1740
|
+
re.I,
|
|
1741
|
+
):
|
|
1742
|
+
return True
|
|
1743
|
+
if hard and text.startswith("Destructive 更显式") and "UI Copy Patterns" in chain:
|
|
1744
|
+
return True
|
|
1745
|
+
if hard and "Design quality checks" in chain and re.search(r"\bnot\s+forced\b", text, re.I):
|
|
1746
|
+
return True
|
|
1747
|
+
return False
|
|
1748
|
+
|
|
1749
|
+
|
|
1750
|
+
def _threshold_square_expands(before_term: str, after_terms: Iterable[str]) -> bool:
|
|
1751
|
+
match = re.fullmatch(r"(\d+(?:\.\d+)?)\s*(px|pt|dp|sp)", before_term, re.I)
|
|
1752
|
+
if not match:
|
|
1753
|
+
return False
|
|
1754
|
+
number, unit = match.groups()
|
|
1755
|
+
square = re.compile(
|
|
1756
|
+
rf"^{re.escape(number)}\s*[x×]\s*{re.escape(number)}\s*{re.escape(unit)}$",
|
|
1757
|
+
re.I,
|
|
1758
|
+
)
|
|
1759
|
+
return any(square.fullmatch(term) for term in after_terms)
|
|
1760
|
+
|
|
1761
|
+
|
|
1762
|
+
def _verb_stem(term: str) -> str:
|
|
1763
|
+
folded = term.casefold()
|
|
1764
|
+
for suffix in ("ing", "ed", "es", "s"):
|
|
1765
|
+
if folded.endswith(suffix) and len(folded) > len(suffix) + 2:
|
|
1766
|
+
return folded[: -len(suffix)]
|
|
1767
|
+
return folded
|
|
1768
|
+
|
|
1769
|
+
|
|
1770
|
+
def _precondition_completion_boundary(before: str, after: str) -> bool:
|
|
1771
|
+
"""Prove ``check before closing`` via an ``only after proof`` boundary.
|
|
1772
|
+
|
|
1773
|
+
This is deliberately narrower than treating ``before`` and ``after`` as
|
|
1774
|
+
interchangeable. The source must make closing conditional on a check,
|
|
1775
|
+
while the carrier must prohibit leaving/closing the contract until proof
|
|
1776
|
+
has established and checked the complete consumer boundary.
|
|
1777
|
+
"""
|
|
1778
|
+
return bool(
|
|
1779
|
+
re.search(r"\bcheck\b.*\bbefore\s+closing\b", before, re.I)
|
|
1780
|
+
and re.search(r"\bonly\s+after\b", after, re.I)
|
|
1781
|
+
and re.search(r"\b(?:establish(?:es|ed)?|proof|proven)\b", after, re.I)
|
|
1782
|
+
and re.search(r"\bcheck(?:ed|s|ing)?\b", after, re.I)
|
|
1783
|
+
)
|
|
1784
|
+
|
|
1785
|
+
|
|
1786
|
+
def _invalid_closure_hard_prohibition(before: str, after: str) -> bool:
|
|
1787
|
+
"""Prove an invalid closure using an explicit close/leave prohibition."""
|
|
1788
|
+
return bool(
|
|
1789
|
+
re.search(r"\bclosure\b.*\binvalid\b", before, re.I)
|
|
1790
|
+
and (
|
|
1791
|
+
re.search(
|
|
1792
|
+
r"\b(?:do\s+not|must\s+not|cannot|never)\b[^.]{0,240}"
|
|
1793
|
+
r"\b(?:close|leave)\b",
|
|
1794
|
+
after,
|
|
1795
|
+
re.I,
|
|
1796
|
+
)
|
|
1797
|
+
or re.search(
|
|
1798
|
+
r"\bonly\b[^.]{0,240}\bproven\b[^.]{0,240}\bmay\s+leave\b",
|
|
1799
|
+
after,
|
|
1800
|
+
re.I,
|
|
1801
|
+
)
|
|
1802
|
+
or re.search(
|
|
1803
|
+
r"\boutside\b[^.]{0,240}\bonly\s+after\b[^.]{0,240}"
|
|
1804
|
+
r"\b(?:check(?:ed|s|ing)?|proves?)\b",
|
|
1805
|
+
after,
|
|
1806
|
+
re.I,
|
|
1807
|
+
)
|
|
1808
|
+
)
|
|
1809
|
+
)
|
|
1810
|
+
|
|
1811
|
+
|
|
1812
|
+
def ensure_partition_qualifier_strength(
|
|
1813
|
+
before: str,
|
|
1814
|
+
after: str,
|
|
1815
|
+
label: str,
|
|
1816
|
+
resolutions: Any,
|
|
1817
|
+
carriers: Iterable[Carrier],
|
|
1818
|
+
) -> None:
|
|
1819
|
+
"""Require every source qualifier literally in a multi-carrier part.
|
|
1820
|
+
|
|
1821
|
+
Opposite-polarity words may legitimately occur in another member that
|
|
1822
|
+
states the bounded exception. Unlike a scalar rewrite, a closed bundle is
|
|
1823
|
+
therefore rejected on a missing source qualifier, not merely because a
|
|
1824
|
+
second member also names the exception's opposite polarity.
|
|
1825
|
+
"""
|
|
1826
|
+
required = qualifier_terms(before)
|
|
1827
|
+
available = qualifier_terms(after)
|
|
1828
|
+
if not isinstance(resolutions, list):
|
|
1829
|
+
raise AuditError("PART_QUALIFIER_RESOLUTIONS_INVALID", label)
|
|
1830
|
+
missing = {
|
|
1831
|
+
(kind, term.casefold()): term
|
|
1832
|
+
for kind, terms in required.items()
|
|
1833
|
+
for term in terms
|
|
1834
|
+
if term.casefold() not in {value.casefold() for value in available.get(kind, [])}
|
|
1835
|
+
}
|
|
1836
|
+
claimed: set[tuple[str, str]] = set()
|
|
1837
|
+
for item in resolutions:
|
|
1838
|
+
if not isinstance(item, dict) or set(item) != _PART_QUALIFIER_RESOLUTION_FIELDS:
|
|
1839
|
+
raise AuditError("PART_QUALIFIER_RESOLUTION_INVALID", label)
|
|
1840
|
+
kind = item.get("kind")
|
|
1841
|
+
before_term = item.get("before")
|
|
1842
|
+
resolution = item.get("resolution")
|
|
1843
|
+
if not isinstance(kind, str) or not isinstance(before_term, str) or not isinstance(resolution, str):
|
|
1844
|
+
raise AuditError("PART_QUALIFIER_RESOLUTION_INVALID", label)
|
|
1845
|
+
key = (kind, before_term.casefold())
|
|
1846
|
+
if key not in missing or key in claimed:
|
|
1847
|
+
raise AuditError("PART_QUALIFIER_RESOLUTION_INVALID", f"{label}: {kind}={before_term}")
|
|
1848
|
+
ok = False
|
|
1849
|
+
if kind == "modality" and resolution == "implicit-normative":
|
|
1850
|
+
ok = _implicit_normative_strength(before_term, carriers)
|
|
1851
|
+
elif kind == "scope" and before_term.casefold() == "per" and resolution == "universal-member":
|
|
1852
|
+
ok = bool(
|
|
1853
|
+
{"any", "all", "each", "every"}
|
|
1854
|
+
& {value.casefold() for value in available.get("scope", [])}
|
|
1855
|
+
)
|
|
1856
|
+
elif kind == "actor" and resolution == "singular-plural":
|
|
1857
|
+
forms = {value.casefold() for value in available.get("actor", [])}
|
|
1858
|
+
folded = before_term.casefold()
|
|
1859
|
+
ok = (folded.rstrip("s") in {value.rstrip("s") for value in forms})
|
|
1860
|
+
elif kind == "threshold" and resolution == "square-dimension":
|
|
1861
|
+
ok = _threshold_square_expands(before_term, available.get("threshold", []))
|
|
1862
|
+
elif kind == "consequence" and resolution == "hard-prohibition":
|
|
1863
|
+
if before_term.casefold() in {"fail", "fails", "block", "blocks"}:
|
|
1864
|
+
ok = bool(HARD_MODALITY.search(after))
|
|
1865
|
+
elif before_term.casefold() == "invalid":
|
|
1866
|
+
ok = _invalid_closure_hard_prohibition(before, after)
|
|
1867
|
+
elif kind == "consequence" and resolution == "verb-inflection":
|
|
1868
|
+
stem = _verb_stem(before_term)
|
|
1869
|
+
ok = any(_verb_stem(term) == stem for term in available.get("consequence", []))
|
|
1870
|
+
elif kind == "recency" and before_term.casefold() == "after" and resolution == "revision-boundary":
|
|
1871
|
+
ok = bool(re.search(r"\b(?:until|revised|revision|fresh)\b", after, re.I))
|
|
1872
|
+
elif kind == "recency" and before_term.casefold() == "before" and resolution == "precondition-completion-boundary":
|
|
1873
|
+
ok = _precondition_completion_boundary(before, after)
|
|
1874
|
+
if not ok:
|
|
1875
|
+
raise AuditError("PART_QUALIFIER_RESOLUTION_INVALID", f"{label}: {kind}={before_term}")
|
|
1876
|
+
claimed.add(key)
|
|
1877
|
+
if claimed != set(missing):
|
|
1878
|
+
unresolved = sorted(f"{kind}={term}" for (kind, _), term in missing.items() if (kind, term.casefold()) not in claimed)
|
|
1879
|
+
raise AuditError("PART_QUALIFIER_UNRESOLVED", f"{label}: {','.join(unresolved)}")
|
|
1880
|
+
missing_classes = sorted(
|
|
1881
|
+
kind for kind in required if kind not in available and not any(k == kind for k, _ in claimed)
|
|
1882
|
+
)
|
|
1883
|
+
if missing_classes:
|
|
1884
|
+
raise AuditError(
|
|
1885
|
+
"PART_QUALIFIER_MISSING",
|
|
1886
|
+
f"{label}: classes={','.join(missing_classes)}",
|
|
1887
|
+
)
|
|
1888
|
+
for kind, terms in required.items():
|
|
1889
|
+
after_terms = {term.casefold() for term in available.get(kind, [])}
|
|
1890
|
+
absent = sorted(
|
|
1891
|
+
term
|
|
1892
|
+
for term in terms
|
|
1893
|
+
if term.casefold() not in after_terms and (kind, term.casefold()) not in claimed
|
|
1894
|
+
)
|
|
1895
|
+
if absent:
|
|
1896
|
+
raise AuditError(
|
|
1897
|
+
"PART_QUALIFIER_LITERAL_MISSING",
|
|
1898
|
+
f"{label}: {kind}={','.join(absent)}",
|
|
1899
|
+
)
|
|
1900
|
+
if HARD_MODALITY.search(before) and not HARD_MODALITY.search(after) and not any(
|
|
1901
|
+
kind == "modality" for kind, _ in claimed
|
|
1902
|
+
):
|
|
1903
|
+
raise AuditError("PART_QUALIFIER_WEAKENED", label)
|
|
1904
|
+
|
|
1905
|
+
|
|
1906
|
+
def validate_partitioned(
|
|
1907
|
+
repo: Path, expected: ExpectedRow, mapping: dict[str, Any]
|
|
1908
|
+
) -> PartitionedCarrier:
|
|
1909
|
+
label = f"{expected.source_path}#{expected.source_ordinal}"
|
|
1910
|
+
if mapping.get("manual_reviewed") is not True:
|
|
1911
|
+
raise AuditError("MANUAL_REVIEW_REQUIRED", f"{label}: schema4 partition")
|
|
1912
|
+
validate_semantic_review(mapping, label, required=True)
|
|
1913
|
+
parts = mapping.get("parts")
|
|
1914
|
+
if not isinstance(parts, list) or not parts:
|
|
1915
|
+
raise AuditError("PARTS_REQUIRED", label)
|
|
1916
|
+
|
|
1917
|
+
cursor = 0
|
|
1918
|
+
seen_ids: set[str] = set()
|
|
1919
|
+
statuses: list[str] = []
|
|
1920
|
+
part_effects: list[str] = []
|
|
1921
|
+
resolved: PartitionedCarrier = []
|
|
1922
|
+
survive_fields = {
|
|
1923
|
+
"id",
|
|
1924
|
+
"status",
|
|
1925
|
+
"source_start",
|
|
1926
|
+
"source_end",
|
|
1927
|
+
"source_text",
|
|
1928
|
+
"effect",
|
|
1929
|
+
"carriers",
|
|
1930
|
+
"qualifiers",
|
|
1931
|
+
"qualifier_resolutions",
|
|
1932
|
+
"semantic_review",
|
|
1933
|
+
"semantic_rationale",
|
|
1934
|
+
}
|
|
1935
|
+
retired_fields = {
|
|
1936
|
+
"id",
|
|
1937
|
+
"status",
|
|
1938
|
+
"source_start",
|
|
1939
|
+
"source_end",
|
|
1940
|
+
"source_text",
|
|
1941
|
+
"authority",
|
|
1942
|
+
"scope",
|
|
1943
|
+
"reason",
|
|
1944
|
+
"semantic_review",
|
|
1945
|
+
"semantic_rationale",
|
|
1946
|
+
}
|
|
1947
|
+
carrier_fields = {
|
|
1948
|
+
"carrier_path",
|
|
1949
|
+
"carrier_text",
|
|
1950
|
+
"carrier_chain",
|
|
1951
|
+
"carrier_sha256",
|
|
1952
|
+
"bridge_terms",
|
|
1953
|
+
}
|
|
1954
|
+
live_source_union = " ".join(
|
|
1955
|
+
str(part.get("source_text", ""))
|
|
1956
|
+
for part in parts
|
|
1957
|
+
if isinstance(part, dict) and part.get("status") == "survives"
|
|
1958
|
+
)
|
|
1959
|
+
for index, part in enumerate(parts, 1):
|
|
1960
|
+
part_label = f"{label}/part-{index}"
|
|
1961
|
+
if not isinstance(part, dict):
|
|
1962
|
+
raise AuditError("PART_FIELDS", part_label)
|
|
1963
|
+
status = part.get("status")
|
|
1964
|
+
allowed = survive_fields if status == "survives" else retired_fields if status == "retired" else set()
|
|
1965
|
+
if not allowed or set(part) != allowed:
|
|
1966
|
+
raise AuditError("PART_FIELDS", part_label)
|
|
1967
|
+
part_id = part.get("id")
|
|
1968
|
+
start = part.get("source_start")
|
|
1969
|
+
end = part.get("source_end")
|
|
1970
|
+
source_text = part.get("source_text")
|
|
1971
|
+
if (
|
|
1972
|
+
not isinstance(part_id, str)
|
|
1973
|
+
or not part_id.strip()
|
|
1974
|
+
or part_id in seen_ids
|
|
1975
|
+
or not isinstance(start, int)
|
|
1976
|
+
or isinstance(start, bool)
|
|
1977
|
+
or not isinstance(end, int)
|
|
1978
|
+
or isinstance(end, bool)
|
|
1979
|
+
or start != cursor
|
|
1980
|
+
or end <= start
|
|
1981
|
+
or end > len(expected.before_text)
|
|
1982
|
+
or not isinstance(source_text, str)
|
|
1983
|
+
or not source_text.strip()
|
|
1984
|
+
):
|
|
1985
|
+
raise AuditError("PARTITION_GAP_OR_OVERLAP", part_label)
|
|
1986
|
+
if not _span_boundary_ok(expected.before_text, start) or not _span_boundary_ok(
|
|
1987
|
+
expected.before_text, end
|
|
1988
|
+
):
|
|
1989
|
+
raise AuditError("PARTITION_SPLITS_TOKEN", part_label)
|
|
1990
|
+
if expected.before_text[start:end] != source_text:
|
|
1991
|
+
raise AuditError("PART_SOURCE_MISMATCH", part_label)
|
|
1992
|
+
cursor = end
|
|
1993
|
+
seen_ids.add(part_id)
|
|
1994
|
+
statuses.append(status)
|
|
1995
|
+
validate_semantic_review(part, part_label, required=True)
|
|
1996
|
+
|
|
1997
|
+
if status == "retired":
|
|
1998
|
+
for field in ("authority", "scope", "reason"):
|
|
1999
|
+
value = part.get(field)
|
|
2000
|
+
if not isinstance(value, str) or not value.strip():
|
|
2001
|
+
raise AuditError("PART_RETIREMENT_PROOF_MISSING", f"{part_label}: {field}")
|
|
2002
|
+
resolved.append(PartitionedPart(part_id, status, source_text, ()))
|
|
2003
|
+
continue
|
|
2004
|
+
|
|
2005
|
+
effect = part.get("effect")
|
|
2006
|
+
if effect not in {"preserved", "strengthened"}:
|
|
2007
|
+
raise AuditError("PART_EFFECT_INVALID", part_label)
|
|
2008
|
+
part_effects.append(str(effect))
|
|
2009
|
+
members = part.get("carriers")
|
|
2010
|
+
if not isinstance(members, list) or not members:
|
|
2011
|
+
raise AuditError("PART_CARRIER_REQUIRED", part_label)
|
|
2012
|
+
part_carriers: list[Carrier] = []
|
|
2013
|
+
part_composites: set[tuple[str, tuple[str, ...], str]] = set()
|
|
2014
|
+
for member_index, member in enumerate(members, 1):
|
|
2015
|
+
member_label = f"{part_label}/member-{member_index}"
|
|
2016
|
+
if not isinstance(member, dict) or set(member) != carrier_fields:
|
|
2017
|
+
raise AuditError("PART_CARRIER_FIELDS", member_label)
|
|
2018
|
+
path = member.get("carrier_path")
|
|
2019
|
+
text = member.get("carrier_text")
|
|
2020
|
+
chain = member.get("carrier_chain")
|
|
2021
|
+
digest = member.get("carrier_sha256")
|
|
2022
|
+
bridges = member.get("bridge_terms")
|
|
2023
|
+
if (
|
|
2024
|
+
not isinstance(path, str)
|
|
2025
|
+
or not path.strip()
|
|
2026
|
+
or not isinstance(text, str)
|
|
2027
|
+
or not text.strip()
|
|
2028
|
+
or not isinstance(chain, list)
|
|
2029
|
+
or not chain
|
|
2030
|
+
or not isinstance(digest, str)
|
|
2031
|
+
or not re.fullmatch(r"[0-9a-f]{64}", digest)
|
|
2032
|
+
or not isinstance(bridges, list)
|
|
2033
|
+
or not bridges
|
|
2034
|
+
or any(not isinstance(term, str) or not term.strip() for term in bridges)
|
|
2035
|
+
):
|
|
2036
|
+
raise AuditError("PART_CARRIER_EMPTY", member_label)
|
|
2037
|
+
path = normalize_path(path)
|
|
2038
|
+
if Path(path).name in PROVENANCE_BASENAMES:
|
|
2039
|
+
raise AuditError("PROVENANCE_ONLY_CARRIER", member_label)
|
|
2040
|
+
try:
|
|
2041
|
+
carrier = exact_carrier(repo, path, text, chain)
|
|
2042
|
+
except AuditError as exc:
|
|
2043
|
+
raise AuditError(exc.code, f"{member_label}: {exc.detail}") from exc
|
|
2044
|
+
if digest != carrier_digest(carrier.path, carrier.chain, carrier.text):
|
|
2045
|
+
raise AuditError("PART_CARRIER_HASH_MISMATCH", member_label)
|
|
2046
|
+
if not all(
|
|
2047
|
+
_literal_bridge_present(live_source_union, carrier.text, term)
|
|
2048
|
+
for term in bridges
|
|
2049
|
+
):
|
|
2050
|
+
raise AuditError("PART_CARRIER_BRIDGE_MISSING", member_label)
|
|
2051
|
+
composite = (carrier.path, carrier.chain, carrier.text)
|
|
2052
|
+
if composite in part_composites:
|
|
2053
|
+
raise AuditError("PART_CARRIER_DUPLICATE", member_label)
|
|
2054
|
+
part_composites.add(composite)
|
|
2055
|
+
part_carriers.append(carrier)
|
|
2056
|
+
aggregate = " ".join(carrier.text for carrier in part_carriers)
|
|
2057
|
+
validate_bundle_summary_proof(
|
|
2058
|
+
[(source_text, aggregate)], part.get("qualifiers"), part_label
|
|
2059
|
+
)
|
|
2060
|
+
ensure_partition_qualifier_strength(
|
|
2061
|
+
source_text,
|
|
2062
|
+
aggregate,
|
|
2063
|
+
part_label,
|
|
2064
|
+
part.get("qualifier_resolutions"),
|
|
2065
|
+
part_carriers,
|
|
2066
|
+
)
|
|
2067
|
+
resolved.append(
|
|
2068
|
+
PartitionedPart(part_id, status, source_text, tuple(part_carriers))
|
|
2069
|
+
)
|
|
2070
|
+
|
|
2071
|
+
if cursor != len(expected.before_text):
|
|
2072
|
+
raise AuditError("PARTITION_GAP_OR_OVERLAP", f"{label}: end={cursor}")
|
|
2073
|
+
if "survives" not in statuses:
|
|
2074
|
+
raise AuditError("PARTITION_NO_SURVIVOR", label)
|
|
2075
|
+
disposition = mapping.get("disposition")
|
|
2076
|
+
if disposition == "partial-retirement":
|
|
2077
|
+
if set(statuses) != {"survives", "retired"}:
|
|
2078
|
+
raise AuditError("PARTITION_DISPOSITION_MISMATCH", label)
|
|
2079
|
+
elif disposition == "partitioned":
|
|
2080
|
+
if set(statuses) != {"survives"}:
|
|
2081
|
+
raise AuditError("PARTITION_DISPOSITION_MISMATCH", label)
|
|
2082
|
+
else:
|
|
2083
|
+
raise AuditError("PARTITION_DISPOSITION_MISMATCH", label)
|
|
2084
|
+
expected_effect = (
|
|
2085
|
+
"retired"
|
|
2086
|
+
if "retired" in statuses
|
|
2087
|
+
else "strengthened"
|
|
2088
|
+
if "strengthened" in part_effects
|
|
2089
|
+
else "preserved"
|
|
2090
|
+
)
|
|
2091
|
+
if mapping.get("effect") != expected_effect:
|
|
2092
|
+
raise AuditError("PARTITION_EFFECT_MISMATCH", label)
|
|
2093
|
+
return resolved
|
|
2094
|
+
|
|
2095
|
+
|
|
2096
|
+
def validate_bundle(
|
|
2097
|
+
repo: Path, expected: ExpectedRow, mapping: dict[str, Any]
|
|
2098
|
+
) -> BundleCarrier:
|
|
2099
|
+
label = f"{expected.source_path}#{expected.source_ordinal}"
|
|
2100
|
+
if mapping.get("disposition") != "subsumed" or mapping.get("effect") == "unresolved":
|
|
2101
|
+
raise AuditError("BUNDLE_STATUS_INVALID", label)
|
|
2102
|
+
if mapping.get("manual_reviewed") is not True:
|
|
2103
|
+
raise AuditError("MANUAL_REVIEW_REQUIRED", f"{label}: carrier_bundle")
|
|
2104
|
+
if any(mapping.get(field) is not None for field in ("carrier_path", "carrier_text", "carrier_chain")):
|
|
2105
|
+
raise AuditError("CARRIER_SHAPE_CONFLICT", label)
|
|
2106
|
+
|
|
2107
|
+
derived_clauses = compound_clauses(expected.before_text)
|
|
2108
|
+
if len(derived_clauses) < 2:
|
|
2109
|
+
raise AuditError("BUNDLE_SOURCE_NOT_COMPOUND", label)
|
|
2110
|
+
if mapping.get("compound_clauses") != derived_clauses:
|
|
2111
|
+
raise AuditError("BUNDLE_CLAUSE_SET_MISMATCH", label)
|
|
2112
|
+
|
|
2113
|
+
bundle = mapping.get("carrier_bundle")
|
|
2114
|
+
if not isinstance(bundle, list) or len(bundle) < 2:
|
|
2115
|
+
raise AuditError("BUNDLE_SIZE", label)
|
|
2116
|
+
allowed_member_fields = {
|
|
2117
|
+
"carrier_path",
|
|
2118
|
+
"carrier_text",
|
|
2119
|
+
"carrier_chain",
|
|
2120
|
+
"covers",
|
|
2121
|
+
"clause_qualifiers",
|
|
2122
|
+
}
|
|
2123
|
+
clause_by_id = {clause["id"]: clause for clause in derived_clauses}
|
|
2124
|
+
coverage: Counter[str] = Counter()
|
|
2125
|
+
resolved: BundleCarrier = []
|
|
2126
|
+
assigned: list[tuple[str, Carrier, str]] = []
|
|
2127
|
+
composites: set[tuple[str, tuple[str, ...], str]] = set()
|
|
2128
|
+
for member_index, member in enumerate(bundle, 1):
|
|
2129
|
+
member_label = f"{label}/member-{member_index}"
|
|
2130
|
+
if not isinstance(member, dict) or set(member) != allowed_member_fields:
|
|
2131
|
+
raise AuditError("BUNDLE_MEMBER_FIELDS", member_label)
|
|
2132
|
+
path = member.get("carrier_path")
|
|
2133
|
+
text = member.get("carrier_text")
|
|
2134
|
+
chain = member.get("carrier_chain")
|
|
2135
|
+
covers = member.get("covers")
|
|
2136
|
+
clause_proofs = member.get("clause_qualifiers")
|
|
2137
|
+
if (
|
|
2138
|
+
not isinstance(path, str)
|
|
2139
|
+
or not path.strip()
|
|
2140
|
+
or not isinstance(text, str)
|
|
2141
|
+
or not text.strip()
|
|
2142
|
+
or not isinstance(chain, list)
|
|
2143
|
+
or not chain
|
|
2144
|
+
or not isinstance(covers, list)
|
|
2145
|
+
or not covers
|
|
2146
|
+
or any(not isinstance(value, str) or not value for value in covers)
|
|
2147
|
+
or not isinstance(clause_proofs, list)
|
|
2148
|
+
):
|
|
2149
|
+
raise AuditError("BUNDLE_MEMBER_EMPTY", member_label)
|
|
2150
|
+
if len(covers) != 1:
|
|
2151
|
+
raise AuditError("BUNDLE_MEMBER_MULTI_CLAUSE", member_label)
|
|
2152
|
+
path = normalize_path(path)
|
|
2153
|
+
if Path(path).name in PROVENANCE_BASENAMES:
|
|
2154
|
+
raise AuditError("PROVENANCE_ONLY_CARRIER", member_label)
|
|
2155
|
+
if compound_clauses(text):
|
|
2156
|
+
raise AuditError("BUNDLE_MEMBER_COMPOUND", member_label)
|
|
2157
|
+
try:
|
|
2158
|
+
carrier = exact_carrier(repo, path, text, chain)
|
|
2159
|
+
except AuditError as exc:
|
|
2160
|
+
raise AuditError(exc.code, f"{member_label}: {exc.detail}") from exc
|
|
2161
|
+
composite = (carrier.path, carrier.chain, carrier.text)
|
|
2162
|
+
if composite in composites:
|
|
2163
|
+
raise AuditError("BUNDLE_DUPLICATE_MEMBER", member_label)
|
|
2164
|
+
composites.add(composite)
|
|
2165
|
+
for clause_id in covers:
|
|
2166
|
+
if clause_id not in clause_by_id:
|
|
2167
|
+
raise AuditError("BUNDLE_UNKNOWN_CLAUSE", f"{member_label}: {clause_id}")
|
|
2168
|
+
coverage[clause_id] += 1
|
|
2169
|
+
assigned.append((clause_id, carrier, member_label))
|
|
2170
|
+
proof_by_clause: dict[str, Any] = {}
|
|
2171
|
+
for proof in clause_proofs:
|
|
2172
|
+
if (
|
|
2173
|
+
not isinstance(proof, dict)
|
|
2174
|
+
or set(proof)
|
|
2175
|
+
!= {
|
|
2176
|
+
"clause_id",
|
|
2177
|
+
"qualifiers",
|
|
2178
|
+
"semantic_review",
|
|
2179
|
+
"semantic_rationale",
|
|
2180
|
+
}
|
|
2181
|
+
or not isinstance(proof.get("clause_id"), str)
|
|
2182
|
+
or proof["clause_id"] in proof_by_clause
|
|
2183
|
+
):
|
|
2184
|
+
raise AuditError("BUNDLE_CLAUSE_PROOF_INVALID", member_label)
|
|
2185
|
+
proof_by_clause[proof["clause_id"]] = proof
|
|
2186
|
+
if set(proof_by_clause) != set(covers):
|
|
2187
|
+
raise AuditError("BUNDLE_CLAUSE_PROOF_INVALID", member_label)
|
|
2188
|
+
for clause_id in covers:
|
|
2189
|
+
proof = proof_by_clause[clause_id]
|
|
2190
|
+
clause_label = f"{member_label}/{clause_id}"
|
|
2191
|
+
validate_semantic_review(proof, clause_label, required=True)
|
|
2192
|
+
validate_bundle_clause_proof(
|
|
2193
|
+
clause_by_id[clause_id]["text"],
|
|
2194
|
+
carrier.text,
|
|
2195
|
+
proof.get("qualifiers"),
|
|
2196
|
+
clause_label,
|
|
2197
|
+
)
|
|
2198
|
+
resolved.append((carrier, tuple(covers)))
|
|
2199
|
+
|
|
2200
|
+
duplicates = sorted(clause_id for clause_id, count in coverage.items() if count > 1)
|
|
2201
|
+
if duplicates:
|
|
2202
|
+
raise AuditError("BUNDLE_CLAUSE_DUPLICATE", f"{label}: {','.join(duplicates)}")
|
|
2203
|
+
missing = sorted(set(clause_by_id) - set(coverage))
|
|
2204
|
+
if missing:
|
|
2205
|
+
raise AuditError("BUNDLE_CLAUSE_UNCOVERED", f"{label}: {','.join(missing)}")
|
|
2206
|
+
|
|
2207
|
+
assignment_by_clause = {
|
|
2208
|
+
clause_id: carrier for clause_id, carrier, _ in assigned
|
|
2209
|
+
}
|
|
2210
|
+
validate_bundle_summary_proof(
|
|
2211
|
+
(
|
|
2212
|
+
(clause["text"], assignment_by_clause[clause["id"]].text)
|
|
2213
|
+
for clause in derived_clauses
|
|
2214
|
+
),
|
|
2215
|
+
mapping.get("qualifiers"),
|
|
2216
|
+
label,
|
|
2217
|
+
)
|
|
2218
|
+
return resolved
|
|
2219
|
+
|
|
2220
|
+
|
|
2221
|
+
def validate_mapping(
|
|
2222
|
+
repo: Path,
|
|
2223
|
+
expected: list[ExpectedRow],
|
|
2224
|
+
mapping_rows: list[dict[str, Any]],
|
|
2225
|
+
*,
|
|
2226
|
+
allow_unresolved: bool,
|
|
2227
|
+
) -> tuple[
|
|
2228
|
+
dict[tuple[str, int], Carrier | BundleCarrier | PartitionedCarrier | None],
|
|
2229
|
+
list[str],
|
|
2230
|
+
]:
|
|
2231
|
+
indexed = index_mapping(mapping_rows)
|
|
2232
|
+
verify_closed_row_set(expected, indexed)
|
|
2233
|
+
resolved: dict[
|
|
2234
|
+
tuple[str, int], Carrier | BundleCarrier | PartitionedCarrier | None
|
|
2235
|
+
] = {}
|
|
2236
|
+
unresolved: list[str] = []
|
|
2237
|
+
|
|
2238
|
+
for source in expected:
|
|
2239
|
+
row = indexed[source.key]
|
|
2240
|
+
label = f"{source.source_path}#{source.source_ordinal}"
|
|
2241
|
+
schema_version = row.get("schema_version")
|
|
2242
|
+
if schema_version not in SCHEMA_VERSIONS:
|
|
2243
|
+
raise AuditError("SCHEMA_VERSION", label)
|
|
2244
|
+
allowed_fields = MAPPING_FIELDS_V4 if schema_version == 4 else MAPPING_FIELDS
|
|
2245
|
+
unknown = sorted(set(row) - allowed_fields)
|
|
2246
|
+
if unknown:
|
|
2247
|
+
if "carriers" in unknown:
|
|
2248
|
+
raise AuditError("MULTIPLE_CARRIERS", label)
|
|
2249
|
+
raise AuditError("UNKNOWN_MAPPING_FIELD", f"{label}: {','.join(unknown)}")
|
|
2250
|
+
if "semantic_review" not in row or "semantic_rationale" not in row:
|
|
2251
|
+
raise AuditError("SEMANTIC_REVIEW_FIELDS_REQUIRED", label)
|
|
2252
|
+
if row.get("disposition") not in DISPOSITIONS:
|
|
2253
|
+
raise AuditError("INVALID_DISPOSITION", label)
|
|
2254
|
+
if row.get("effect") not in EFFECTS:
|
|
2255
|
+
raise AuditError("INVALID_EFFECT", label)
|
|
2256
|
+
if not isinstance(row.get("review_note"), str) or not row["review_note"].strip():
|
|
2257
|
+
raise AuditError("REVIEW_NOTE_MISSING", label)
|
|
2258
|
+
if schema_version == 4:
|
|
2259
|
+
if row.get("disposition") not in {"partitioned", "partial-retirement"}:
|
|
2260
|
+
raise AuditError("PARTITION_DISPOSITION_MISMATCH", label)
|
|
2261
|
+
resolved[source.key] = validate_partitioned(repo, source, row)
|
|
2262
|
+
continue
|
|
2263
|
+
if row.get("disposition") in {"partitioned", "partial-retirement"}:
|
|
2264
|
+
raise AuditError("SCHEMA_VERSION", f"{label}: partition requires schema 4")
|
|
2265
|
+
relations = row.get("qualifier_relations", [])
|
|
2266
|
+
if not isinstance(relations, list):
|
|
2267
|
+
raise AuditError("QUALIFIER_RELATION_INVALID", f"{label}: array required")
|
|
2268
|
+
if relations and row["effect"] != "strengthened":
|
|
2269
|
+
raise AuditError("QUALIFIER_RELATION_REQUIRES_STRENGTHENED", label)
|
|
2270
|
+
if row["disposition"] == "retired-dead" and row["effect"] not in {
|
|
2271
|
+
"retired",
|
|
2272
|
+
"unresolved",
|
|
2273
|
+
}:
|
|
2274
|
+
raise AuditError("RETIRED_EFFECT_INVALID", label)
|
|
2275
|
+
if row["effect"] == "retired" and row["disposition"] != "retired-dead":
|
|
2276
|
+
raise AuditError("RETIRED_EFFECT_INVALID", label)
|
|
2277
|
+
if row["effect"] == "strengthened" and row.get("manual_reviewed") is not True:
|
|
2278
|
+
raise AuditError("STRENGTHENED_REVIEW_REQUIRED", label)
|
|
2279
|
+
if relations and (
|
|
2280
|
+
row["disposition"] == "retired-dead"
|
|
2281
|
+
or row["effect"] == "unresolved"
|
|
2282
|
+
or row.get("carrier_bundle") is not None
|
|
2283
|
+
):
|
|
2284
|
+
raise AuditError("QUALIFIER_RELATION_FORBIDDEN", label)
|
|
2285
|
+
|
|
2286
|
+
if row["effect"] == "unresolved":
|
|
2287
|
+
validate_semantic_review(row, label, required=False)
|
|
2288
|
+
unresolved.append(label)
|
|
2289
|
+
if any(
|
|
2290
|
+
row.get(field) is not None
|
|
2291
|
+
for field in (
|
|
2292
|
+
"carrier_path",
|
|
2293
|
+
"carrier_text",
|
|
2294
|
+
"carrier_chain",
|
|
2295
|
+
"carrier_bundle",
|
|
2296
|
+
"compound_clauses",
|
|
2297
|
+
)
|
|
2298
|
+
):
|
|
2299
|
+
raise AuditError("UNRESOLVED_HAS_CARRIER", label)
|
|
2300
|
+
resolved[source.key] = None
|
|
2301
|
+
continue
|
|
2302
|
+
|
|
2303
|
+
if row["disposition"] == "retired-dead":
|
|
2304
|
+
if any(
|
|
2305
|
+
row.get(field) is not None
|
|
2306
|
+
for field in (
|
|
2307
|
+
"carrier_path",
|
|
2308
|
+
"carrier_text",
|
|
2309
|
+
"carrier_chain",
|
|
2310
|
+
"carrier_bundle",
|
|
2311
|
+
"compound_clauses",
|
|
2312
|
+
)
|
|
2313
|
+
):
|
|
2314
|
+
raise AuditError("RETIRED_HAS_CARRIER", label)
|
|
2315
|
+
if row.get("manual_reviewed") is not True:
|
|
2316
|
+
raise AuditError("MANUAL_REVIEW_REQUIRED", label)
|
|
2317
|
+
validate_semantic_review(row, label, required=True)
|
|
2318
|
+
note = row["review_note"].casefold()
|
|
2319
|
+
if "authority:" not in note or "scope:" not in note:
|
|
2320
|
+
raise AuditError("RETIREMENT_PROOF_MISSING", label)
|
|
2321
|
+
resolved[source.key] = None
|
|
2322
|
+
continue
|
|
2323
|
+
|
|
2324
|
+
|
|
2325
|
+
if row.get("carrier_bundle") is not None:
|
|
2326
|
+
validate_semantic_review(row, label, required=False)
|
|
2327
|
+
resolved[source.key] = validate_bundle(repo, source, row)
|
|
2328
|
+
continue
|
|
2329
|
+
if row.get("compound_clauses") is not None:
|
|
2330
|
+
raise AuditError("CARRIER_SHAPE_CONFLICT", label)
|
|
2331
|
+
|
|
2332
|
+
carrier_path = row.get("carrier_path")
|
|
2333
|
+
carrier_text = row.get("carrier_text")
|
|
2334
|
+
carrier_chain = row.get("carrier_chain")
|
|
2335
|
+
if any(isinstance(value, list) for value in (carrier_path, carrier_text)) or (
|
|
2336
|
+
isinstance(carrier_chain, list) and carrier_chain and isinstance(carrier_chain[0], list)
|
|
2337
|
+
):
|
|
2338
|
+
raise AuditError("MULTIPLE_CARRIERS", label)
|
|
2339
|
+
if not isinstance(carrier_path, str) or not isinstance(carrier_text, str) or not isinstance(carrier_chain, list):
|
|
2340
|
+
raise AuditError("CARRIER_REQUIRED", label)
|
|
2341
|
+
carrier_path = normalize_path(carrier_path)
|
|
2342
|
+
if Path(carrier_path).name in PROVENANCE_BASENAMES:
|
|
2343
|
+
raise AuditError("PROVENANCE_ONLY_CARRIER", label)
|
|
2344
|
+
try:
|
|
2345
|
+
carrier = exact_carrier(repo, carrier_path, carrier_text, carrier_chain)
|
|
2346
|
+
except AuditError as exc:
|
|
2347
|
+
raise AuditError(exc.code, f"{label}: {exc.detail}") from exc
|
|
2348
|
+
if row["disposition"] in {"merged", "subsumed"} and carrier.chain != source.before_chain:
|
|
2349
|
+
raise AuditError("DISPOSITION_CHAIN_CONFLICT", f"{label}: same-host disposition moved")
|
|
2350
|
+
if row["disposition"] == "rehosted" and carrier.chain == source.before_chain:
|
|
2351
|
+
raise AuditError("DISPOSITION_CHAIN_CONFLICT", f"{label}: rehosted without host change")
|
|
2352
|
+
validate_semantic_review(
|
|
2353
|
+
row,
|
|
2354
|
+
label,
|
|
2355
|
+
required=GCD.normalize(source.before_text) != carrier.text,
|
|
2356
|
+
)
|
|
2357
|
+
validate_qualifiers(source, row, carrier)
|
|
2358
|
+
resolved[source.key] = carrier
|
|
2359
|
+
|
|
2360
|
+
if unresolved and not allow_unresolved:
|
|
2361
|
+
raise AuditError(
|
|
2362
|
+
"UNRESOLVED_ROWS",
|
|
2363
|
+
f"count={len(unresolved)} first={','.join(unresolved[:8])}",
|
|
2364
|
+
)
|
|
2365
|
+
return resolved, unresolved
|
|
2366
|
+
|
|
2367
|
+
|
|
2368
|
+
def md(value: str) -> str:
|
|
2369
|
+
return value.replace("|", "\\|").replace("\n", " ")
|
|
2370
|
+
|
|
2371
|
+
|
|
2372
|
+
def chain_text(value: Iterable[str]) -> str:
|
|
2373
|
+
items = list(value)
|
|
2374
|
+
return " → ".join(md(item) for item in items) if items else "(root)"
|
|
2375
|
+
|
|
2376
|
+
|
|
2377
|
+
def proof_mode(
|
|
2378
|
+
source: ExpectedRow,
|
|
2379
|
+
row: dict[str, Any],
|
|
2380
|
+
carrier: Carrier | BundleCarrier | PartitionedCarrier | None,
|
|
2381
|
+
) -> str:
|
|
2382
|
+
if carrier is None and row["effect"] == "unresolved":
|
|
2383
|
+
return "unresolved"
|
|
2384
|
+
if isinstance(carrier, Carrier) and GCD.normalize(source.before_text) == carrier.text:
|
|
2385
|
+
return "exact-mechanical"
|
|
2386
|
+
return "reviewed-semantic"
|
|
2387
|
+
|
|
2388
|
+
|
|
2389
|
+
def proof_mode_counts(
|
|
2390
|
+
expected: Iterable[ExpectedRow],
|
|
2391
|
+
mapping_rows: list[dict[str, Any]],
|
|
2392
|
+
resolved: dict[
|
|
2393
|
+
tuple[str, int], Carrier | BundleCarrier | PartitionedCarrier | None
|
|
2394
|
+
],
|
|
2395
|
+
) -> Counter[str]:
|
|
2396
|
+
indexed = index_mapping(mapping_rows)
|
|
2397
|
+
return Counter(
|
|
2398
|
+
proof_mode(source, indexed[source.key], resolved[source.key])
|
|
2399
|
+
for source in expected
|
|
2400
|
+
)
|
|
2401
|
+
|
|
2402
|
+
|
|
2403
|
+
def render_ledger(
|
|
2404
|
+
base: str,
|
|
2405
|
+
head: str | None,
|
|
2406
|
+
comparison_paths: list[str],
|
|
2407
|
+
relocation_paths: list[str],
|
|
2408
|
+
expected: list[ExpectedRow],
|
|
2409
|
+
mapping_rows: list[dict[str, Any]],
|
|
2410
|
+
resolved: dict[
|
|
2411
|
+
tuple[str, int], Carrier | BundleCarrier | PartitionedCarrier | None
|
|
2412
|
+
],
|
|
2413
|
+
unresolved: list[str],
|
|
2414
|
+
) -> str:
|
|
2415
|
+
indexed = index_mapping(mapping_rows)
|
|
2416
|
+
effects = Counter(str(indexed[row.key]["effect"]) for row in expected)
|
|
2417
|
+
dispositions = Counter(str(indexed[row.key]["disposition"]) for row in expected)
|
|
2418
|
+
part_counts: Counter[str] = Counter()
|
|
2419
|
+
for row in expected:
|
|
2420
|
+
for part in indexed[row.key].get("parts") or []:
|
|
2421
|
+
if part.get("status") == "retired":
|
|
2422
|
+
part_counts["retired"] += 1
|
|
2423
|
+
else:
|
|
2424
|
+
part_counts[f"survived-{part.get('effect')}"] += 1
|
|
2425
|
+
modes = proof_mode_counts(expected, mapping_rows, resolved)
|
|
2426
|
+
source_counts: dict[str, Counter[str]] = {}
|
|
2427
|
+
for source in expected:
|
|
2428
|
+
row = indexed[source.key]
|
|
2429
|
+
counts = source_counts.setdefault(source.source_path, Counter())
|
|
2430
|
+
counts["rows"] += 1
|
|
2431
|
+
counts[str(row["effect"])] += 1
|
|
2432
|
+
counts[str(row["disposition"])] += 1
|
|
2433
|
+
lines = [
|
|
2434
|
+
"# UI/UX obligation-preservation ledger",
|
|
2435
|
+
"",
|
|
2436
|
+
"> Generated by `skills/skill-extraction-workflow/scripts/obligation-ledger.py`; do not hand-edit.",
|
|
2437
|
+
"> The sibling `obligation-mapping.jsonl` is the canonical proof source; this compact Markdown file is its generated reader index.",
|
|
2438
|
+
"> Candidate binding is intentionally external and must be written only after the worktree is frozen.",
|
|
2439
|
+
"",
|
|
2440
|
+
"## Reproducible comparison domain",
|
|
2441
|
+
"",
|
|
2442
|
+
f"- Base revision: `{base}`",
|
|
2443
|
+
*([f"- Head revision: `{head}`"] if head else []),
|
|
2444
|
+
f"- Changed pre-existing `skills/**/*.md`: {len(comparison_paths)}",
|
|
2445
|
+
f"- Explicit relocation destinations: {len(relocation_paths)}",
|
|
2446
|
+
f"- Governing-chain-diff rows: {len(expected)}",
|
|
2447
|
+
f"- Effects: preserved={effects['preserved']}, strengthened={effects['strengthened']}, retired={effects['retired']}, unresolved={len(unresolved)}",
|
|
2448
|
+
"- Partition parts (partitioned/partial-retirement rows close row-level as their dominant effect; part-level outcomes are counted here so neither the deleted nor the strengthened dimension of a mixed row disappears): "
|
|
2449
|
+
f"survived-preserved={part_counts['survived-preserved']}, "
|
|
2450
|
+
f"survived-strengthened={part_counts['survived-strengthened']}, "
|
|
2451
|
+
f"retired={part_counts['retired']}",
|
|
2452
|
+
"- Proof modes: "
|
|
2453
|
+
f"exact-mechanical={modes['exact-mechanical']}, "
|
|
2454
|
+
f"reviewed-semantic={modes['reviewed-semantic']}, "
|
|
2455
|
+
f"unresolved={modes['unresolved']}",
|
|
2456
|
+
"- Dispositions: " + ", ".join(
|
|
2457
|
+
f"{name}={dispositions[name]}" for name in sorted(dispositions)
|
|
2458
|
+
),
|
|
2459
|
+
"",
|
|
2460
|
+
"The row set is the bidirectional equality of the mechanical governing-chain diff and the JSONL mapping manifest. Exact before/carrier text, lexical qualifier evidence, semantic-review decisions, and review notes live only in the canonical mapping. `proof_mode=reviewed-semantic` validates review evidence presence and shape, not the truth or correctness of the semantic judgment. Carrier line ranges below are recomputed from exact current text, so stale locators still fail audit.",
|
|
2461
|
+
"",
|
|
2462
|
+
"## Source summary",
|
|
2463
|
+
"",
|
|
2464
|
+
"| Source | Rows | Preserved | Strengthened | Retired |",
|
|
2465
|
+
"| --- | ---: | ---: | ---: | ---: |",
|
|
2466
|
+
]
|
|
2467
|
+
for source_path, counts in sorted(source_counts.items()):
|
|
2468
|
+
lines.append(
|
|
2469
|
+
f"| `{source_path}` | {counts['rows']} | {counts['preserved']} | {counts['strengthened']} | {counts['retired']} |"
|
|
2470
|
+
)
|
|
2471
|
+
lines.extend(
|
|
2472
|
+
[
|
|
2473
|
+
"",
|
|
2474
|
+
"## Compact obligation index",
|
|
2475
|
+
"",
|
|
2476
|
+
"| ID | Source row | Reason | Decision | Exact current locator / bundle coverage | Proof index |",
|
|
2477
|
+
"| --- | --- | --- | --- | --- | --- |",
|
|
2478
|
+
]
|
|
2479
|
+
)
|
|
2480
|
+
for serial, source in enumerate(expected, 1):
|
|
2481
|
+
row = indexed[source.key]
|
|
2482
|
+
carrier = resolved[source.key]
|
|
2483
|
+
if carrier is None:
|
|
2484
|
+
carrier_label = "—"
|
|
2485
|
+
elif isinstance(carrier, list) and carrier and isinstance(
|
|
2486
|
+
carrier[0], PartitionedPart
|
|
2487
|
+
):
|
|
2488
|
+
labels = []
|
|
2489
|
+
for part in carrier:
|
|
2490
|
+
if part.status == "retired":
|
|
2491
|
+
labels.append(f"{part.part_id}. retired")
|
|
2492
|
+
else:
|
|
2493
|
+
labels.extend(
|
|
2494
|
+
f"{part.part_id}.{index}. `{item.path}:{item.start_line}-{item.end_line}`"
|
|
2495
|
+
for index, item in enumerate(part.carriers, 1)
|
|
2496
|
+
)
|
|
2497
|
+
carrier_label = "<br>".join(labels)
|
|
2498
|
+
elif isinstance(carrier, list):
|
|
2499
|
+
carrier_label = "<br>".join(
|
|
2500
|
+
f"{index}. `{item.path}:{item.start_line}-{item.end_line}` [{','.join(covers)}]"
|
|
2501
|
+
for index, (item, covers) in enumerate(carrier, 1)
|
|
2502
|
+
)
|
|
2503
|
+
else:
|
|
2504
|
+
carrier_label = f"`{carrier.path}:{carrier.start_line}-{carrier.end_line}`"
|
|
2505
|
+
qualifiers = row.get("qualifiers") or []
|
|
2506
|
+
relation_count = len(row.get("qualifier_relations") or [])
|
|
2507
|
+
mode = proof_mode(source, row, carrier)
|
|
2508
|
+
if isinstance(carrier, list) and carrier and isinstance(
|
|
2509
|
+
carrier[0], PartitionedPart
|
|
2510
|
+
):
|
|
2511
|
+
survived = sum(part.status == "survives" for part in carrier)
|
|
2512
|
+
retired = sum(part.status == "retired" for part in carrier)
|
|
2513
|
+
members = sum(len(part.carriers) for part in carrier)
|
|
2514
|
+
proof_label = (
|
|
2515
|
+
f"proof_mode={mode}; schema4 exact-span partition; "
|
|
2516
|
+
f"survives={survived}; retired={retired}; carriers={members}; "
|
|
2517
|
+
"hash+bridge+qualifier checked"
|
|
2518
|
+
)
|
|
2519
|
+
elif isinstance(carrier, list):
|
|
2520
|
+
proof_label = (
|
|
2521
|
+
f"proof_mode={mode}; reviewed closed bundle; members={len(carrier)}; "
|
|
2522
|
+
"clause-local lexical guards"
|
|
2523
|
+
)
|
|
2524
|
+
elif carrier and GCD.normalize(source.before_text) == carrier.text:
|
|
2525
|
+
proof_label = f"proof_mode={mode}; verbatim exact carrier"
|
|
2526
|
+
elif carrier:
|
|
2527
|
+
kinds = ",".join(str(item["kind"]) for item in qualifiers) or "none"
|
|
2528
|
+
proof_label = (
|
|
2529
|
+
f"proof_mode={mode}; reviewed scalar; lexical qualifier kinds={kinds}"
|
|
2530
|
+
)
|
|
2531
|
+
elif row["effect"] == "unresolved":
|
|
2532
|
+
proof_label = f"proof_mode={mode}"
|
|
2533
|
+
else:
|
|
2534
|
+
proof_label = (
|
|
2535
|
+
f"proof_mode={mode}; reviewed retirement; "
|
|
2536
|
+
"authority+scope evidence in canonical mapping"
|
|
2537
|
+
)
|
|
2538
|
+
if relation_count:
|
|
2539
|
+
proof_label += f"; relations={relation_count}"
|
|
2540
|
+
lines.append(
|
|
2541
|
+
"| "
|
|
2542
|
+
+ " | ".join(
|
|
2543
|
+
[
|
|
2544
|
+
f"O{serial:04d}",
|
|
2545
|
+
f"`{source.source_path}#{source.source_ordinal}`",
|
|
2546
|
+
source.reason,
|
|
2547
|
+
f"{row['disposition']} / {row['effect']}",
|
|
2548
|
+
carrier_label,
|
|
2549
|
+
md(proof_label),
|
|
2550
|
+
]
|
|
2551
|
+
)
|
|
2552
|
+
+ " |"
|
|
2553
|
+
)
|
|
2554
|
+
lines.extend(
|
|
2555
|
+
[
|
|
2556
|
+
"",
|
|
2557
|
+
"## Candidate binding boundary",
|
|
2558
|
+
"",
|
|
2559
|
+
"This document contains no self-referential dirty-worktree digest. After freeze, bind the candidate in a checkout-external manifest and audit this generated ledger against that frozen tree.",
|
|
2560
|
+
"",
|
|
2561
|
+
]
|
|
2562
|
+
)
|
|
2563
|
+
return "\n".join(lines)
|
|
2564
|
+
|
|
2565
|
+
|
|
2566
|
+
def exact_candidates(repo: Path) -> dict[str, list[tuple[str, tuple[str, ...]]]]:
|
|
2567
|
+
candidates: dict[str, list[tuple[str, tuple[str, ...]]]] = {}
|
|
2568
|
+
for file_path in sorted((repo / "skills").rglob("*.md")):
|
|
2569
|
+
relative = file_path.relative_to(repo).as_posix()
|
|
2570
|
+
if file_path.name in PROVENANCE_BASENAMES:
|
|
2571
|
+
continue
|
|
2572
|
+
for obligation, _, _ in parse_obligation_ranges(
|
|
2573
|
+
file_path.read_text(encoding="utf-8")
|
|
2574
|
+
):
|
|
2575
|
+
candidates.setdefault(GCD.normalize(obligation.text), []).append(
|
|
2576
|
+
(relative, tuple(obligation.chain))
|
|
2577
|
+
)
|
|
2578
|
+
return candidates
|
|
2579
|
+
|
|
2580
|
+
|
|
2581
|
+
def skeleton(repo: Path, expected: Iterable[ExpectedRow], *, auto_bind_exact: bool) -> str:
|
|
2582
|
+
candidates = exact_candidates(repo) if auto_bind_exact else {}
|
|
2583
|
+
values = []
|
|
2584
|
+
for row in expected:
|
|
2585
|
+
exact = candidates.get(row.before_text, [])
|
|
2586
|
+
mechanically_bound = len(exact) == 1 and exact[0][1] != row.before_chain
|
|
2587
|
+
if mechanically_bound:
|
|
2588
|
+
carrier_path, carrier_chain = exact[0]
|
|
2589
|
+
disposition = "rehosted"
|
|
2590
|
+
effect = "preserved"
|
|
2591
|
+
carrier_text: str | None = row.before_text
|
|
2592
|
+
manual_reviewed = False
|
|
2593
|
+
review_note = "Mechanically preserved verbatim under one exact current (path, chain, text) carrier."
|
|
2594
|
+
else:
|
|
2595
|
+
disposition = "rehosted"
|
|
2596
|
+
effect = "unresolved"
|
|
2597
|
+
carrier_path = None
|
|
2598
|
+
carrier_text = None
|
|
2599
|
+
carrier_chain = None
|
|
2600
|
+
manual_reviewed = False
|
|
2601
|
+
review_note = "Unresolved: exact current carrier and qualifier preservation require review."
|
|
2602
|
+
values.append(
|
|
2603
|
+
json.dumps(
|
|
2604
|
+
{
|
|
2605
|
+
"schema_version": SCHEMA_VERSION,
|
|
2606
|
+
"source_path": row.source_path,
|
|
2607
|
+
"source_ordinal": row.source_ordinal,
|
|
2608
|
+
"reason": row.reason,
|
|
2609
|
+
"before_text": row.before_text,
|
|
2610
|
+
"before_chain": list(row.before_chain),
|
|
2611
|
+
"disposition": disposition,
|
|
2612
|
+
"effect": effect,
|
|
2613
|
+
"carrier_path": carrier_path,
|
|
2614
|
+
"carrier_text": carrier_text,
|
|
2615
|
+
"carrier_chain": list(carrier_chain) if carrier_chain is not None else None,
|
|
2616
|
+
"carrier_bundle": None,
|
|
2617
|
+
"compound_clauses": None,
|
|
2618
|
+
"manual_reviewed": manual_reviewed,
|
|
2619
|
+
"semantic_review": None,
|
|
2620
|
+
"semantic_rationale": None,
|
|
2621
|
+
"qualifiers": [],
|
|
2622
|
+
"qualifier_relations": [],
|
|
2623
|
+
"review_note": review_note,
|
|
2624
|
+
},
|
|
2625
|
+
ensure_ascii=False,
|
|
2626
|
+
separators=(",", ":"),
|
|
2627
|
+
)
|
|
2628
|
+
)
|
|
2629
|
+
return "\n".join(values) + ("\n" if values else "")
|
|
2630
|
+
|
|
2631
|
+
|
|
2632
|
+
def write_atomic(path: Path, content: str) -> None:
|
|
2633
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
2634
|
+
temporary = path.with_name(path.name + ".tmp")
|
|
2635
|
+
temporary.write_text(content, encoding="utf-8")
|
|
2636
|
+
temporary.replace(path)
|
|
2637
|
+
|
|
2638
|
+
|
|
2639
|
+
def main(argv: list[str]) -> int:
|
|
2640
|
+
parser = argparse.ArgumentParser(description=__doc__)
|
|
2641
|
+
parser.add_argument("command", choices=("inventory", "render", "audit"))
|
|
2642
|
+
parser.add_argument("--repo", default=".")
|
|
2643
|
+
parser.add_argument("--base", required=True)
|
|
2644
|
+
parser.add_argument(
|
|
2645
|
+
"--head",
|
|
2646
|
+
help=(
|
|
2647
|
+
"pin the comparison-domain head to this commit; without it the "
|
|
2648
|
+
"domain runs to the working tree, so a repository-frozen ledger "
|
|
2649
|
+
"would demand rows from every later, unrelated change"
|
|
2650
|
+
),
|
|
2651
|
+
)
|
|
2652
|
+
parser.add_argument("--mapping")
|
|
2653
|
+
parser.add_argument("--ledger")
|
|
2654
|
+
parser.add_argument("--output")
|
|
2655
|
+
parser.add_argument(
|
|
2656
|
+
"--no-auto-bind-exact",
|
|
2657
|
+
action="store_true",
|
|
2658
|
+
help="leave even globally unique verbatim rehosts unresolved in inventory output",
|
|
2659
|
+
)
|
|
2660
|
+
args = parser.parse_args(argv)
|
|
2661
|
+
|
|
2662
|
+
repo = Path(args.repo).resolve()
|
|
2663
|
+
try:
|
|
2664
|
+
# A movable ref (origin/dev) is not a reproducible baseline: resolve it
|
|
2665
|
+
# to a commit ONCE and derive everything — header included — from that
|
|
2666
|
+
# commit, so the rendered header, the comparison domain, and the rows
|
|
2667
|
+
# cannot disagree even if the ref moves mid-run. The header records
|
|
2668
|
+
# ONLY the resolved SHA, so a render invoked via any ref spelling is
|
|
2669
|
+
# byte-identical and audit's byte comparison turns silent baseline
|
|
2670
|
+
# drift into an explicit STALE_LEDGER failure.
|
|
2671
|
+
resolved_base = git(repo, "rev-parse", f"{args.base}^{{commit}}").stdout.strip()
|
|
2672
|
+
resolved_head = (
|
|
2673
|
+
git(repo, "rev-parse", f"{args.head}^{{commit}}").stdout.strip()
|
|
2674
|
+
if args.head
|
|
2675
|
+
else None
|
|
2676
|
+
)
|
|
2677
|
+
comparison = changed_preexisting_paths(repo, resolved_base, resolved_head)
|
|
2678
|
+
expected = derive_rows(repo, resolved_base, comparison, resolved_head)
|
|
2679
|
+
if args.command == "inventory":
|
|
2680
|
+
content = skeleton(repo, expected, auto_bind_exact=not args.no_auto_bind_exact)
|
|
2681
|
+
if args.output:
|
|
2682
|
+
write_atomic(Path(args.output), content)
|
|
2683
|
+
else:
|
|
2684
|
+
sys.stdout.write(content)
|
|
2685
|
+
print(
|
|
2686
|
+
f"inventory_ok domain={len(comparison)} rows={len(expected)}",
|
|
2687
|
+
file=sys.stderr,
|
|
2688
|
+
)
|
|
2689
|
+
return 0
|
|
2690
|
+
|
|
2691
|
+
if not args.mapping:
|
|
2692
|
+
raise AuditError("MAPPING_REQUIRED", "--mapping")
|
|
2693
|
+
mapping_path = Path(args.mapping)
|
|
2694
|
+
mapping_rows = load_mapping(mapping_path)
|
|
2695
|
+
relocations = relocation_destinations(mapping_rows)
|
|
2696
|
+
# Relocation destinations belong to the comparison domain even if they
|
|
2697
|
+
# are new or unchanged; only pre-existing changed paths can owe rows.
|
|
2698
|
+
domain = sorted(set(comparison) | set(relocations))
|
|
2699
|
+
resolved, unresolved = validate_mapping(
|
|
2700
|
+
repo,
|
|
2701
|
+
expected,
|
|
2702
|
+
mapping_rows,
|
|
2703
|
+
allow_unresolved=args.command == "render",
|
|
2704
|
+
)
|
|
2705
|
+
modes = proof_mode_counts(expected, mapping_rows, resolved)
|
|
2706
|
+
rendered = render_ledger(
|
|
2707
|
+
resolved_base,
|
|
2708
|
+
resolved_head,
|
|
2709
|
+
comparison,
|
|
2710
|
+
sorted(set(relocations) - set(comparison)),
|
|
2711
|
+
expected,
|
|
2712
|
+
mapping_rows,
|
|
2713
|
+
resolved,
|
|
2714
|
+
unresolved,
|
|
2715
|
+
)
|
|
2716
|
+
if args.command == "render":
|
|
2717
|
+
if not args.output:
|
|
2718
|
+
sys.stdout.write(rendered)
|
|
2719
|
+
else:
|
|
2720
|
+
write_atomic(Path(args.output), rendered)
|
|
2721
|
+
print(
|
|
2722
|
+
f"render_ok domain={len(domain)} rows={len(expected)} "
|
|
2723
|
+
f"unresolved={len(unresolved)} "
|
|
2724
|
+
f"exact_mechanical={modes['exact-mechanical']} "
|
|
2725
|
+
f"reviewed_semantic={modes['reviewed-semantic']}",
|
|
2726
|
+
file=sys.stderr,
|
|
2727
|
+
)
|
|
2728
|
+
return 0
|
|
2729
|
+
|
|
2730
|
+
if not args.ledger:
|
|
2731
|
+
raise AuditError("LEDGER_REQUIRED", "--ledger")
|
|
2732
|
+
ledger = Path(args.ledger)
|
|
2733
|
+
if not ledger.is_file() or ledger.read_text(encoding="utf-8") != rendered:
|
|
2734
|
+
raise AuditError("STALE_LEDGER", str(ledger))
|
|
2735
|
+
print(
|
|
2736
|
+
f"audit_ok domain={len(domain)} rows={len(expected)} unresolved=0 "
|
|
2737
|
+
f"exact_mechanical={modes['exact-mechanical']} "
|
|
2738
|
+
f"reviewed_semantic={modes['reviewed-semantic']}",
|
|
2739
|
+
file=sys.stderr,
|
|
2740
|
+
)
|
|
2741
|
+
return 0
|
|
2742
|
+
except AuditError as exc:
|
|
2743
|
+
print(f"ERROR {exc.code}: {exc.detail}", file=sys.stderr)
|
|
2744
|
+
return 1
|
|
2745
|
+
|
|
2746
|
+
|
|
2747
|
+
if __name__ == "__main__":
|
|
2748
|
+
raise SystemExit(main(sys.argv[1:]))
|