@ictechgy/context-guard 0.4.14 → 0.4.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +7 -0
- package/README.ko.md +26 -0
- package/README.md +27 -0
- package/docs/benchmark-fixtures/image-context-pack-full-evidence.prompt.example.md +28 -0
- package/docs/benchmark-fixtures/image-context-pack-packed-evidence.prompt.example.md +31 -0
- package/docs/benchmark-fixtures/image-context-pack.evidence.example.jsonl +2 -0
- package/docs/benchmark-fixtures/image-context-pack.tasks.example.json +18 -0
- package/docs/benchmark-fixtures/image-context-pack.variants.example.json +10 -0
- package/docs/benchmark-workflow-examples.md +16 -0
- package/docs/experimental-benchmark-fixtures.md +52 -1
- package/package.json +2 -1
- package/plugins/context-guard/.claude-plugin/plugin.json +1 -1
- package/plugins/context-guard/README.ko.md +25 -0
- package/plugins/context-guard/README.md +25 -0
- package/plugins/context-guard/bin/context-guard-bench +1305 -115
- package/plugins/context-guard/bin/context-guard-experiments +3080 -133
- package/plugins/context-guard/bin/context-guard-mcp +999 -0
- package/plugins/context-guard/bin/context-guard-pack +634 -9
- package/plugins/context-guard/lib/context_guard_commands.py +8 -0
|
@@ -11,6 +11,7 @@ from __future__ import annotations
|
|
|
11
11
|
import argparse
|
|
12
12
|
from dataclasses import asdict, dataclass
|
|
13
13
|
from datetime import datetime, timezone
|
|
14
|
+
import errno
|
|
14
15
|
import http.client
|
|
15
16
|
from http.server import BaseHTTPRequestHandler, HTTPServer
|
|
16
17
|
import hashlib
|
|
@@ -26,6 +27,7 @@ import shlex
|
|
|
26
27
|
import socket
|
|
27
28
|
from socketserver import TCPServer
|
|
28
29
|
from pathlib import Path
|
|
30
|
+
import unicodedata
|
|
29
31
|
import stat
|
|
30
32
|
import sys
|
|
31
33
|
import time
|
|
@@ -70,6 +72,237 @@ LOCAL_PROXY_EXTERNAL_DESIGN_SCHEMA_VERSION = "contextguard.experiments.local-pro
|
|
|
70
72
|
LOCAL_PROXY_RESPONSE_SANDBOX_SCHEMA_VERSION = "contextguard.experiments.local-proxy-response-sandbox.v1"
|
|
71
73
|
IMAGE_CONTEXT_PACK_PLAN_SCHEMA_VERSION = "contextguard.experiments.image-context-pack-plan.v1"
|
|
72
74
|
SEMANTIC_CHECKPOINT_PLAN_SCHEMA_VERSION = "contextguard.experiments.semantic-checkpoint-plan.v1"
|
|
75
|
+
PROOF_CARRYING_CONTEXT_PLAN_SCHEMA_VERSION = "contextguard.experiments.proof-carrying-context-plan.v1"
|
|
76
|
+
PROOF_CARRYING_CONTEXT_VERIFY_SCHEMA_VERSION = "contextguard.experiments.proof-carrying-context-verification.v1"
|
|
77
|
+
PROOF_CARRYING_CONTEXT_UNIT_SCHEMA_VERSION = "contextguard.proof-unit.v1"
|
|
78
|
+
PROOF_CARRYING_CONTEXT_DETAILED_UNIT_CAP = 64
|
|
79
|
+
PROOF_CARRYING_CONTEXT_UNIT_JSON_BYTE_CAP = 8192
|
|
80
|
+
PROOF_UNIT_JSON_MAX_DEPTH = 100
|
|
81
|
+
PROOF_RECEIPT_METADATA_BYTE_CAP = 64_000
|
|
82
|
+
PROOF_RECEIPT_CONTENT_BYTE_CAP = 100_000_000
|
|
83
|
+
PROOF_RECEIPT_CONTENT_READ_CHUNK = 1_048_576
|
|
84
|
+
SEMANTIC_GC_PLAN_SCHEMA_VERSION = "contextguard.experiments.semantic-gc-plan.v1"
|
|
85
|
+
SEMANTIC_GC_UNIT_SCHEMA_VERSION = "contextguard.semantic-gc-unit.v1"
|
|
86
|
+
SEMANTIC_GC_DETAILED_UNIT_CAP = 64
|
|
87
|
+
SEMANTIC_GC_UNIT_JSON_BYTE_CAP = 8192
|
|
88
|
+
SEMANTIC_GC_JSON_MAX_DEPTH = 100
|
|
89
|
+
SEMANTIC_GC_PROCESS_EXIT_CONTRACT = (
|
|
90
|
+
"exit code 0 means ready_for_plan_review; exit code 2 means a blocked plan was emitted"
|
|
91
|
+
)
|
|
92
|
+
STATIC_RELEVANCE_PLAN_SCHEMA_VERSION = "contextguard.experiments.static-relevance-plan.v1"
|
|
93
|
+
STATIC_RELEVANCE_UNIT_SCHEMA_VERSION = "contextguard.static-relevance-unit.v1"
|
|
94
|
+
STATIC_RELEVANCE_DETAILED_UNIT_CAP = 64
|
|
95
|
+
STATIC_RELEVANCE_UNIT_JSON_BYTE_CAP = 8192
|
|
96
|
+
STATIC_RELEVANCE_JSON_MAX_DEPTH = 100
|
|
97
|
+
STATIC_RELEVANCE_PROCESS_EXIT_CONTRACT = (
|
|
98
|
+
"exit code 0 means ready_for_plan_review; exit code 2 means a blocked plan was emitted"
|
|
99
|
+
)
|
|
100
|
+
JSON_SAFE_INTEGER_MAX = 9_007_199_254_740_991
|
|
101
|
+
PROOF_SOURCE_LABEL_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._:/+-]{0,119}$")
|
|
102
|
+
PROOF_RECEIPT_ID_RE = re.compile(r"^[a-f0-9]{16,64}$")
|
|
103
|
+
PROOF_CONTENT_SHA256_RE = re.compile(r"^[a-f0-9]{64}$")
|
|
104
|
+
PROOF_CAPTURED_AT_RE = re.compile(r"^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}Z$")
|
|
105
|
+
PROOF_UNIT_ALLOWED_FIELDS = frozenset({
|
|
106
|
+
"source_label",
|
|
107
|
+
"receipt_id",
|
|
108
|
+
"content_sha256",
|
|
109
|
+
"safe_range",
|
|
110
|
+
"captured_at",
|
|
111
|
+
"transform_policy",
|
|
112
|
+
"rehydrate_command",
|
|
113
|
+
})
|
|
114
|
+
PROOF_READINESS_BLOCKER_ORDER = (
|
|
115
|
+
"missing_proof_unit",
|
|
116
|
+
"too_many_proof_units",
|
|
117
|
+
"proof_unit_json_too_large",
|
|
118
|
+
"invalid_proof_unit_unicode",
|
|
119
|
+
"invalid_proof_unit_json",
|
|
120
|
+
"duplicate_proof_unit_keys",
|
|
121
|
+
"proof_unit_json_nesting_too_deep",
|
|
122
|
+
"nonfinite_proof_unit_number",
|
|
123
|
+
"proof_unit_not_object",
|
|
124
|
+
"unknown_proof_unit_fields",
|
|
125
|
+
"missing_source_label",
|
|
126
|
+
"invalid_source_label",
|
|
127
|
+
"missing_receipt",
|
|
128
|
+
"invalid_receipt",
|
|
129
|
+
"missing_content_sha256",
|
|
130
|
+
"invalid_content_sha256",
|
|
131
|
+
"missing_timestamp",
|
|
132
|
+
"invalid_timestamp",
|
|
133
|
+
"missing_transform_policy",
|
|
134
|
+
"invalid_transform_policy",
|
|
135
|
+
"invalid_safe_range",
|
|
136
|
+
"missing_safe_range_for_transform_policy",
|
|
137
|
+
"missing_rehydrate_command",
|
|
138
|
+
"invalid_rehydrate_command",
|
|
139
|
+
"rehydrate_receipt_mismatch",
|
|
140
|
+
"receipt_hash_conflict",
|
|
141
|
+
"protected_zone_denial_required",
|
|
142
|
+
"missing_provider_measurement_boundary",
|
|
143
|
+
)
|
|
144
|
+
PROOF_WARNING_ORDER = (
|
|
145
|
+
"protected_zone_compliance_not_checked",
|
|
146
|
+
"safe_range_bounds_not_checked",
|
|
147
|
+
"receipt_storage_not_checked",
|
|
148
|
+
"content_hash_not_verified",
|
|
149
|
+
"rehydrate_command_not_executed",
|
|
150
|
+
"timestamp_freshness_not_checked",
|
|
151
|
+
"safe_range_omitted",
|
|
152
|
+
"duplicate_proof_unit",
|
|
153
|
+
)
|
|
154
|
+
PROOF_VERIFICATION_BLOCKER_ORDER = (
|
|
155
|
+
"missing_proof_unit",
|
|
156
|
+
"too_many_proof_units",
|
|
157
|
+
"proof_unit_json_too_large",
|
|
158
|
+
"invalid_proof_unit_unicode",
|
|
159
|
+
"invalid_proof_unit_json",
|
|
160
|
+
"duplicate_proof_unit_keys",
|
|
161
|
+
"proof_unit_json_nesting_too_deep",
|
|
162
|
+
"nonfinite_proof_unit_number",
|
|
163
|
+
"proof_unit_not_object",
|
|
164
|
+
"unknown_proof_unit_fields",
|
|
165
|
+
"missing_source_label",
|
|
166
|
+
"invalid_source_label",
|
|
167
|
+
"missing_receipt",
|
|
168
|
+
"invalid_receipt",
|
|
169
|
+
"missing_content_sha256",
|
|
170
|
+
"invalid_content_sha256",
|
|
171
|
+
"missing_timestamp",
|
|
172
|
+
"invalid_timestamp",
|
|
173
|
+
"missing_transform_policy",
|
|
174
|
+
"invalid_transform_policy",
|
|
175
|
+
"invalid_safe_range",
|
|
176
|
+
"missing_safe_range_for_transform_policy",
|
|
177
|
+
"missing_rehydrate_command",
|
|
178
|
+
"invalid_rehydrate_command",
|
|
179
|
+
"rehydrate_receipt_mismatch",
|
|
180
|
+
"receipt_hash_conflict",
|
|
181
|
+
"invalid_artifact_directory",
|
|
182
|
+
"artifact_io_capability_unavailable",
|
|
183
|
+
"artifact_directory_not_found",
|
|
184
|
+
"artifact_directory_symlink_rejected",
|
|
185
|
+
"artifact_directory_not_regular",
|
|
186
|
+
"artifact_directory_owner_mismatch",
|
|
187
|
+
"artifact_directory_mode_not_private",
|
|
188
|
+
"artifact_directory_access_failed",
|
|
189
|
+
"request_preflight_aborted",
|
|
190
|
+
"receipt_pair_incomplete",
|
|
191
|
+
"receipt_metadata_symlink_rejected",
|
|
192
|
+
"receipt_metadata_not_regular",
|
|
193
|
+
"receipt_metadata_owner_mismatch",
|
|
194
|
+
"receipt_metadata_mode_not_private",
|
|
195
|
+
"receipt_metadata_multiple_links",
|
|
196
|
+
"receipt_metadata_too_large",
|
|
197
|
+
"receipt_metadata_invalid_unicode",
|
|
198
|
+
"receipt_metadata_invalid_json",
|
|
199
|
+
"receipt_metadata_duplicate_keys",
|
|
200
|
+
"receipt_metadata_nesting_too_deep",
|
|
201
|
+
"receipt_metadata_nonfinite_number",
|
|
202
|
+
"receipt_metadata_not_object",
|
|
203
|
+
"receipt_metadata_id_mismatch",
|
|
204
|
+
"receipt_metadata_stored_output_invalid",
|
|
205
|
+
"receipt_metadata_file_binding_mismatch",
|
|
206
|
+
"receipt_content_symlink_rejected",
|
|
207
|
+
"receipt_content_not_regular",
|
|
208
|
+
"receipt_content_owner_mismatch",
|
|
209
|
+
"receipt_content_mode_not_private",
|
|
210
|
+
"receipt_content_multiple_links",
|
|
211
|
+
"receipt_content_too_large",
|
|
212
|
+
"receipt_content_size_mismatch",
|
|
213
|
+
"receipt_content_hash_mismatch",
|
|
214
|
+
"receipt_line_count_mismatch",
|
|
215
|
+
"proof_content_hash_mismatch",
|
|
216
|
+
"safe_range_out_of_bounds",
|
|
217
|
+
"artifact_changed_during_read",
|
|
218
|
+
"artifact_read_failed",
|
|
219
|
+
)
|
|
220
|
+
PROOF_VERIFICATION_WARNING_ORDER = (
|
|
221
|
+
"timestamp_freshness_not_checked",
|
|
222
|
+
"protected_zone_compliance_not_checked",
|
|
223
|
+
"rehydrate_command_not_executed",
|
|
224
|
+
"safe_range_not_supplied",
|
|
225
|
+
"duplicate_proof_unit",
|
|
226
|
+
)
|
|
227
|
+
PROOF_VERIFICATION_CLAIM_BOUNDARY = (
|
|
228
|
+
"Local receipt/hash/range/command binding only; no semantic-safety, protected-zone, freshness, replacement, "
|
|
229
|
+
"omission, or hosted-savings authority."
|
|
230
|
+
)
|
|
231
|
+
PROOF_VERIFICATION_PROCESS_EXIT_CONTRACT = (
|
|
232
|
+
"exit code 0 means all supplied proof units passed bounded local verification only; exit code 2 means "
|
|
233
|
+
"verification_failed"
|
|
234
|
+
)
|
|
235
|
+
SEMANTIC_GC_UNIT_ID_RE = re.compile(r"^[A-Za-z0-9._:/-]{1,128}$")
|
|
236
|
+
SEMANTIC_GC_SOURCE_LABEL_RE = re.compile(r"^[A-Za-z0-9._:/ -]{1,128}$")
|
|
237
|
+
SEMANTIC_GC_RECEIPT_ID_RE = re.compile(r"^[a-f0-9]{16,64}$")
|
|
238
|
+
SEMANTIC_GC_CONTENT_SHA256_RE = re.compile(r"^[a-f0-9]{64}$")
|
|
239
|
+
SEMANTIC_GC_ALLOWED_FIELDS = frozenset({
|
|
240
|
+
"schema", "unit_id", "references", "is_root", "protected_zone",
|
|
241
|
+
"content_sha256", "provenance", "missed_context_note", "exact_fallback_command",
|
|
242
|
+
})
|
|
243
|
+
SEMANTIC_GC_BLOCKER_ORDER = (
|
|
244
|
+
"no_context_units", "unit_limit_exceeded", "invalid_context_unit_json",
|
|
245
|
+
"duplicate_json_key", "context_unit_depth_exceeded", "nonfinite_json_number",
|
|
246
|
+
"invalid_unicode_scalar", "decoder_recursion_limit", "invalid_context_unit_schema",
|
|
247
|
+
"unknown_context_unit_field", "missing_unit_id", "invalid_unit_id",
|
|
248
|
+
"duplicate_unit_id", "invalid_references", "duplicate_reference", "unknown_reference",
|
|
249
|
+
"ambiguous_reference",
|
|
250
|
+
"invalid_root_flag", "invalid_protected_zone_flag", "no_declared_root",
|
|
251
|
+
"graph_evaluation_suppressed", "protected_zone_policy_required",
|
|
252
|
+
"invalid_content_sha256", "missing_provenance", "invalid_provenance",
|
|
253
|
+
"invalid_source_label", "invalid_receipt_id", "missing_missed_context_note",
|
|
254
|
+
"invalid_missed_context_note", "missing_exact_fallback", "invalid_exact_fallback",
|
|
255
|
+
"fallback_receipt_mismatch", "provider_boundary_ack_required", "human_review_ack_required",
|
|
256
|
+
)
|
|
257
|
+
SEMANTIC_GC_WARNING_ORDER = (
|
|
258
|
+
"plan_only_no_omission", "caller_declared_graph_unverified",
|
|
259
|
+
"semantic_relevance_not_evaluated", "provider_boundary_not_verified",
|
|
260
|
+
"provenance_not_verified_externally", "fallback_not_executed",
|
|
261
|
+
"human_review_still_required", "accepted_notes_are_untrusted",
|
|
262
|
+
"duplicate_content_sha256", "duplicate_receipt_id",
|
|
263
|
+
"protected_unreachable_excluded", "no_sweep_candidates",
|
|
264
|
+
)
|
|
265
|
+
STATIC_RELEVANCE_ID_RE = re.compile(r"^[A-Za-z0-9._:/-]{1,128}$")
|
|
266
|
+
STATIC_RELEVANCE_PATH_TOKEN_SPLIT_RE = re.compile(r"[._-]+")
|
|
267
|
+
STATIC_RELEVANCE_ALLOWED_FIELDS = frozenset({
|
|
268
|
+
"schema", "unit_id", "path", "task_anchor", "protection_reasons", "symbol",
|
|
269
|
+
"symbol_references", "dataflow_predecessors", "dataflow_successors", "git",
|
|
270
|
+
})
|
|
271
|
+
STATIC_RELEVANCE_SYMBOL_KINDS = frozenset({
|
|
272
|
+
"module", "class", "function", "method", "variable", "constant", "test", "config", "data", "unknown",
|
|
273
|
+
})
|
|
274
|
+
STATIC_RELEVANCE_PROTECTION_REASON_ORDER = (
|
|
275
|
+
"authentication", "authorization", "secrets", "security_sensitive", "migration",
|
|
276
|
+
"acceptance_test", "unresolved_error_evidence", "caller_protected",
|
|
277
|
+
"builtin_auth_path", "builtin_security_path", "builtin_secret_path",
|
|
278
|
+
"builtin_migration_path", "builtin_acceptance_path", "builtin_secret_material_path",
|
|
279
|
+
)
|
|
280
|
+
STATIC_RELEVANCE_EXPLICIT_PROTECTION_REASONS = frozenset(STATIC_RELEVANCE_PROTECTION_REASON_ORDER[:8])
|
|
281
|
+
STATIC_RELEVANCE_BLOCKER_ORDER = (
|
|
282
|
+
"no_relevance_units", "relevance_unit_limit_exceeded", "relevance_unit_json_too_large",
|
|
283
|
+
"invalid_unicode_scalar", "decoder_recursion_limit", "malformed_relevance_unit_json",
|
|
284
|
+
"duplicate_relevance_unit_json_key", "non_finite_relevance_unit_json_value",
|
|
285
|
+
"relevance_unit_json_depth_exceeded", "relevance_unit_must_be_object",
|
|
286
|
+
"relevance_unit_schema_mismatch", "relevance_unit_unexpected_field",
|
|
287
|
+
"invalid_relevance_unit_id", "invalid_relevance_unit_path", "invalid_task_anchor",
|
|
288
|
+
"missing_protection_reasons", "invalid_protection_reasons", "duplicate_relevance_unit_id",
|
|
289
|
+
"no_task_anchor", "missing_symbol_signal", "invalid_symbol_signal",
|
|
290
|
+
"missing_symbol_references_signal", "invalid_symbol_references_signal",
|
|
291
|
+
"missing_dataflow_predecessors_signal", "invalid_dataflow_predecessors_signal",
|
|
292
|
+
"missing_dataflow_successors_signal", "invalid_dataflow_successors_signal",
|
|
293
|
+
"missing_git_signal", "invalid_git_signal", "missing_blame_age_signal",
|
|
294
|
+
"invalid_blame_age_signal", "missing_blame_contributor_signal",
|
|
295
|
+
"invalid_blame_contributor_signal", "missing_path_change_count_signal",
|
|
296
|
+
"invalid_path_change_count_signal", "duplicate_relation_target", "unknown_relation_target",
|
|
297
|
+
"ambiguous_relation_target", "inconsistent_dataflow_relation",
|
|
298
|
+
"protected_path_policy_required", "provider_boundary_ack_required",
|
|
299
|
+
)
|
|
300
|
+
STATIC_RELEVANCE_WARNING_ORDER = (
|
|
301
|
+
"caller_declared_static_evidence_unverified", "symbol_and_dataflow_semantics_not_verified",
|
|
302
|
+
"git_history_metrics_not_verified", "protected_path_detection_non_exhaustive",
|
|
303
|
+
"accepted_labels_are_untrusted_caller_data", "static_relevance_is_not_semantic_safety",
|
|
304
|
+
"review_order_does_not_authorize_omission", "hosted_provider_behavior_and_savings_unverified",
|
|
305
|
+
)
|
|
73
306
|
IMAGE_CONTEXT_PACK_PROVIDER_BOUNDARY = "provider-measured-matched-tasks-required"
|
|
74
307
|
LOCAL_PROXY_DEFAULT_BIND_HOST = "127.0.0.1"
|
|
75
308
|
LOCAL_PROXY_DEFAULT_BIND_PORT = 0
|
|
@@ -360,6 +593,128 @@ EXPERIMENTS: tuple[Experiment, ...] = (
|
|
|
360
593
|
"is ready for plan review; raw context remains authoritative."
|
|
361
594
|
),
|
|
362
595
|
),
|
|
596
|
+
Experiment(
|
|
597
|
+
id="proof-carrying-context",
|
|
598
|
+
name="Proof-carrying context metadata planning and local verification gate",
|
|
599
|
+
summary=(
|
|
600
|
+
"Plan proof-envelope metadata readiness or read-only verify explicit private local receipts without "
|
|
601
|
+
"executing rehydration or emitting compact context."
|
|
602
|
+
),
|
|
603
|
+
stability="experimental",
|
|
604
|
+
default_enabled=False,
|
|
605
|
+
risk_level="high",
|
|
606
|
+
claim_boundary=(
|
|
607
|
+
"Local verification covers receipt/content/hash/range bounds and command binding only; it grants no "
|
|
608
|
+
"semantic-safety, freshness, protected-zone, replacement, omission, or hosted-savings authority."
|
|
609
|
+
),
|
|
610
|
+
gate_requirements=(
|
|
611
|
+
"at least one bounded inline proof-unit JSON object",
|
|
612
|
+
"caller-declared protected-zone denial",
|
|
613
|
+
"provider/model measurement boundary acknowledgement",
|
|
614
|
+
"syntax-only proof metadata and exact rehydration binding",
|
|
615
|
+
"one explicit private no-follow artifact directory for read-only verification",
|
|
616
|
+
),
|
|
617
|
+
runtime_status="available-plan-and-read-only-verify",
|
|
618
|
+
commands=(
|
|
619
|
+
"context-guard experiments plan proof-carrying-context",
|
|
620
|
+
"context-guard experiments verify proof-carrying-context",
|
|
621
|
+
),
|
|
622
|
+
opt_in_flags=(
|
|
623
|
+
"plan proof-carrying-context",
|
|
624
|
+
"verify proof-carrying-context",
|
|
625
|
+
"--artifact-dir",
|
|
626
|
+
"--proof-unit-json",
|
|
627
|
+
"--provider-boundary-ack",
|
|
628
|
+
"--protected-zone-policy deny",
|
|
629
|
+
),
|
|
630
|
+
config_effect=(
|
|
631
|
+
"Registry enablement records project-local intent only; proof-carrying-context exposes deterministic plan "
|
|
632
|
+
"and read-only local verify commands. Verify reads only explicit receipt leaves; neither command adds an "
|
|
633
|
+
"emit/record/serve runtime, executes rehydration, writes, generates compact context, or replaces context."
|
|
634
|
+
),
|
|
635
|
+
evidence_contract=(
|
|
636
|
+
"The planner validates bounded inline proof metadata only. Verify additionally checks explicit private "
|
|
637
|
+
"receipt storage, full-content bindings, range bounds, and command syntax without executing it; timestamp "
|
|
638
|
+
"freshness, protected-zone semantics, replacement safety, and hosted savings remain unchecked."
|
|
639
|
+
),
|
|
640
|
+
),
|
|
641
|
+
Experiment(
|
|
642
|
+
id="semantic-gc",
|
|
643
|
+
name="Semantic graph garbage-collection planning gate",
|
|
644
|
+
summary=(
|
|
645
|
+
"Plan-only caller-declared mark-and-sweep classification with strict graph-integrity suppression "
|
|
646
|
+
"and recovery-evidence gates for human review."
|
|
647
|
+
),
|
|
648
|
+
stability="experimental",
|
|
649
|
+
default_enabled=False,
|
|
650
|
+
risk_level="high",
|
|
651
|
+
claim_boundary=(
|
|
652
|
+
"Unreachable graph nodes are plan-review candidates, not proof of irrelevance or authorization to omit; "
|
|
653
|
+
"the planner does not read content, verify provenance, execute fallback, or call providers."
|
|
654
|
+
),
|
|
655
|
+
gate_requirements=(
|
|
656
|
+
"complete unambiguous caller-declared graph",
|
|
657
|
+
"deny-only protected-zone declaration",
|
|
658
|
+
"candidate recovery evidence and missed-context note",
|
|
659
|
+
"provider-boundary acknowledgement for every complete graph; "
|
|
660
|
+
"human-review acknowledgement when unprotected sweep candidates exist",
|
|
661
|
+
),
|
|
662
|
+
runtime_status="available-plan-only",
|
|
663
|
+
commands=("context-guard experiments plan semantic-gc",),
|
|
664
|
+
opt_in_flags=(
|
|
665
|
+
"plan semantic-gc", "--context-unit-json", "--provider-boundary-ack",
|
|
666
|
+
"--human-review-ack", "--protected-zone-policy deny",
|
|
667
|
+
),
|
|
668
|
+
config_effect=(
|
|
669
|
+
"Registry enablement records project-local intent only; semantic-gc exposes one deterministic plan command. "
|
|
670
|
+
"It does not add an emit/record/serve/apply/delete/omit runtime, read context or artifacts, write files, "
|
|
671
|
+
"call models/providers/network, execute fallback, replace context, or authorize omission."
|
|
672
|
+
),
|
|
673
|
+
evidence_contract=(
|
|
674
|
+
"The complete caller-declared graph must pass strict structural validation before iterative reachability. "
|
|
675
|
+
"Unprotected unreachable candidates require sanitized provenance, content hash, exact fallback, and an "
|
|
676
|
+
"untrusted missed-context note plus human-review acknowledgement before the plan is ready for review. "
|
|
677
|
+
"A complete graph with no unprotected sweep candidates does not require that acknowledgement. "
|
|
678
|
+
"Ready plans exit 0; blocked plans still emit their envelope and exit 2."
|
|
679
|
+
),
|
|
680
|
+
),
|
|
681
|
+
Experiment(
|
|
682
|
+
id="static-relevance",
|
|
683
|
+
name="Static relevance evidence compiler",
|
|
684
|
+
summary=(
|
|
685
|
+
"Compile bounded caller-declared symbol, dataflow, and git-history signals into deterministic "
|
|
686
|
+
"human-review diagnostics with protected-retention vetoes."
|
|
687
|
+
),
|
|
688
|
+
stability="experimental",
|
|
689
|
+
default_enabled=False,
|
|
690
|
+
risk_level="high",
|
|
691
|
+
claim_boundary=(
|
|
692
|
+
"Static evidence and review order are unverified plan-review diagnostics only; they do not authorize "
|
|
693
|
+
"deprioritization, omission, deletion, replacement, or runtime action."
|
|
694
|
+
),
|
|
695
|
+
gate_requirements=(
|
|
696
|
+
"complete bounded inline caller-declared static evidence",
|
|
697
|
+
"unambiguous reciprocal dataflow relations and at least one task anchor",
|
|
698
|
+
"deny-only protected-path policy",
|
|
699
|
+
"provider-boundary acknowledgement",
|
|
700
|
+
),
|
|
701
|
+
runtime_status="available-plan-only",
|
|
702
|
+
commands=("context-guard experiments plan static-relevance",),
|
|
703
|
+
opt_in_flags=(
|
|
704
|
+
"plan static-relevance", "--relevance-unit-json", "--provider-boundary-ack",
|
|
705
|
+
"--protected-path-policy deny",
|
|
706
|
+
),
|
|
707
|
+
config_effect=(
|
|
708
|
+
"Registry enablement records project-local intent only; static-relevance exposes one deterministic plan "
|
|
709
|
+
"command and never scans repositories, reads source, invokes git/parsers/providers/subprocesses, writes "
|
|
710
|
+
"files, or authorizes runtime selection or omission."
|
|
711
|
+
),
|
|
712
|
+
evidence_contract=(
|
|
713
|
+
"Every caller-supplied symbol, relation, dataflow, and git signal must be syntactically complete and "
|
|
714
|
+
"internally consistent before slices or review ordering are compiled. Protected paths and explicit "
|
|
715
|
+
"protected evidence are retention vetoes. All evidence remains unverified."
|
|
716
|
+
),
|
|
717
|
+
),
|
|
363
718
|
Experiment(
|
|
364
719
|
id="learned-compression",
|
|
365
720
|
name="Learned/synthetic compression candidate gate",
|
|
@@ -496,6 +851,15 @@ class RegistryError(RuntimeError):
|
|
|
496
851
|
pass
|
|
497
852
|
|
|
498
853
|
|
|
854
|
+
class StoreOnceAction(argparse.Action):
|
|
855
|
+
"""Store a sensitive option once without echoing either supplied value."""
|
|
856
|
+
|
|
857
|
+
def __call__(self, parser, namespace, values, option_string=None):
|
|
858
|
+
if getattr(namespace, self.dest, None) is not None:
|
|
859
|
+
raise argparse.ArgumentError(self, "--artifact-dir may be specified only once")
|
|
860
|
+
setattr(namespace, self.dest, values)
|
|
861
|
+
|
|
862
|
+
|
|
499
863
|
def fail(message: str, code: int = 2) -> NoReturn:
|
|
500
864
|
print(f"{TOOL_NAME}: {message}", file=sys.stderr)
|
|
501
865
|
raise SystemExit(code)
|
|
@@ -1983,159 +2347,2640 @@ def image_context_pack_plan_payload(args: argparse.Namespace) -> dict[str, Any]:
|
|
|
1983
2347
|
}
|
|
1984
2348
|
|
|
1985
2349
|
|
|
1986
|
-
def command_plan_image_context_pack(args: argparse.Namespace) -> int:
|
|
1987
|
-
payload = image_context_pack_plan_payload(args)
|
|
2350
|
+
def command_plan_image_context_pack(args: argparse.Namespace) -> int:
|
|
2351
|
+
payload = image_context_pack_plan_payload(args)
|
|
2352
|
+
if args.json:
|
|
2353
|
+
emit_json(payload)
|
|
2354
|
+
else:
|
|
2355
|
+
print("ContextGuard image-context-pack plan (dry-run only)")
|
|
2356
|
+
print("No image rendering, OCR/image service, model call, proxy forwarding, binary artifact, or replacement was emitted.")
|
|
2357
|
+
print(f"Status: {payload['status']}")
|
|
2358
|
+
if payload["review_plan"]["readiness_blockers"]:
|
|
2359
|
+
print(f"Readiness blockers: {', '.join(payload['review_plan']['readiness_blockers'])}")
|
|
2360
|
+
print(payload["claim_boundary"])
|
|
2361
|
+
return 0
|
|
2362
|
+
|
|
2363
|
+
|
|
2364
|
+
def semantic_checkpoint_plan_payload(args: argparse.Namespace) -> dict[str, Any]:
|
|
2365
|
+
goal = args.goal.strip() if args.goal else None
|
|
2366
|
+
receipt_id = args.exact_context_fallback_receipt.strip() if args.exact_context_fallback_receipt else None
|
|
2367
|
+
reexpand_command = args.reexpand_command.strip() if args.reexpand_command else None
|
|
2368
|
+
reexpand_valid, fallback_blocker = valid_learned_reexpand_command(receipt_id, reexpand_command)
|
|
2369
|
+
fallback_blocker_map = {
|
|
2370
|
+
"missing_exact_fallback": "missing_exact_context_fallback",
|
|
2371
|
+
"invalid_reexpand_command": "invalid_exact_context_reexpand_command",
|
|
2372
|
+
}
|
|
2373
|
+
protected_policy = (args.protected_zone_policy or "deny").strip().lower()
|
|
2374
|
+
missed_context_notes = clean_values(args.missed_context_note)
|
|
2375
|
+
missing_provenance_notes = clean_values(args.missing_provenance_note)
|
|
2376
|
+
|
|
2377
|
+
blockers: list[str] = []
|
|
2378
|
+
if not goal:
|
|
2379
|
+
blockers.append("missing_goal")
|
|
2380
|
+
if fallback_blocker:
|
|
2381
|
+
blockers.append(fallback_blocker_map.get(fallback_blocker, fallback_blocker))
|
|
2382
|
+
if not args.provider_boundary_ack:
|
|
2383
|
+
blockers.append("missing_provider_measurement_boundary")
|
|
2384
|
+
if protected_policy != "deny":
|
|
2385
|
+
blockers.append("protected_zone_denial_required")
|
|
2386
|
+
if not missed_context_notes:
|
|
2387
|
+
blockers.append("missing_missed_context_note")
|
|
2388
|
+
if not missing_provenance_notes:
|
|
2389
|
+
blockers.append("missing_provenance_review")
|
|
2390
|
+
blockers = list(dict.fromkeys(blockers))
|
|
2391
|
+
ready = not blockers
|
|
2392
|
+
|
|
2393
|
+
return {
|
|
2394
|
+
"tool": TOOL_NAME,
|
|
2395
|
+
"schema_version": CONFIG_SCHEMA_VERSION,
|
|
2396
|
+
"plan_schema_version": SEMANTIC_CHECKPOINT_PLAN_SCHEMA_VERSION,
|
|
2397
|
+
"experiment_id": "semantic-checkpoint",
|
|
2398
|
+
"mode": "dry_run",
|
|
2399
|
+
"status": "ready_for_plan_review" if ready else "blocked_until_semantic_checkpoint_gate_ready",
|
|
2400
|
+
"plan_only": {
|
|
2401
|
+
"command_advertised": True,
|
|
2402
|
+
"emit_command_available": False,
|
|
2403
|
+
"record_command_available": False,
|
|
2404
|
+
"serve_command_available": False,
|
|
2405
|
+
"runtime_behavior_changed": False,
|
|
2406
|
+
"replacement_context_emitted": False,
|
|
2407
|
+
},
|
|
2408
|
+
"external_services": {
|
|
2409
|
+
"called": False,
|
|
2410
|
+
"network": False,
|
|
2411
|
+
"model_calls": False,
|
|
2412
|
+
"provider_calls": False,
|
|
2413
|
+
"proxy_forwarding": False,
|
|
2414
|
+
},
|
|
2415
|
+
"runtime_side_effects": {
|
|
2416
|
+
"files_written": False,
|
|
2417
|
+
"transcript_edited": False,
|
|
2418
|
+
"prompt_edited": False,
|
|
2419
|
+
"context_replaced": False,
|
|
2420
|
+
"stable_runtime_behavior_changed": False,
|
|
2421
|
+
},
|
|
2422
|
+
"checkpoint_metadata": {
|
|
2423
|
+
"goal": goal,
|
|
2424
|
+
"constraints": clean_values(args.constraint),
|
|
2425
|
+
"decisions": clean_values(args.decision),
|
|
2426
|
+
"open_tasks": clean_values(args.open_task),
|
|
2427
|
+
"evidence_provenance_handles": clean_values(args.evidence_handle),
|
|
2428
|
+
"unresolved_questions": clean_values(args.unresolved_question),
|
|
2429
|
+
},
|
|
2430
|
+
"exact_context_fallback": {
|
|
2431
|
+
"required": True,
|
|
2432
|
+
"available": bool(reexpand_valid),
|
|
2433
|
+
"receipt_id": receipt_id,
|
|
2434
|
+
"reexpand_command": reexpand_command,
|
|
2435
|
+
"verified": False,
|
|
2436
|
+
"must_be_verified_before_checkpoint_metadata_is_used": True,
|
|
2437
|
+
"allowed_reexpand_shapes": [
|
|
2438
|
+
"context-guard-artifact get RECEIPT --full",
|
|
2439
|
+
"context-guard artifact get RECEIPT --full",
|
|
2440
|
+
],
|
|
2441
|
+
},
|
|
2442
|
+
"protected_zones": {
|
|
2443
|
+
"policy": protected_policy,
|
|
2444
|
+
"override_allowed": False,
|
|
2445
|
+
"denied_classes": [
|
|
2446
|
+
"code",
|
|
2447
|
+
"diffs",
|
|
2448
|
+
"identifiers",
|
|
2449
|
+
"hashes",
|
|
2450
|
+
"paths",
|
|
2451
|
+
"numeric_constants",
|
|
2452
|
+
"json_keys",
|
|
2453
|
+
"stack_frames",
|
|
2454
|
+
"secrets",
|
|
2455
|
+
"prompt_like_instructions",
|
|
2456
|
+
],
|
|
2457
|
+
},
|
|
2458
|
+
"measurement_boundary": {
|
|
2459
|
+
"provider_boundary_acknowledged": bool(args.provider_boundary_ack),
|
|
2460
|
+
"provider_boundary_policy": IMAGE_CONTEXT_PACK_PROVIDER_BOUNDARY,
|
|
2461
|
+
"provider_measured_matched_tasks_required_for_hosted_claims": True,
|
|
2462
|
+
"provider_model_specific": True,
|
|
2463
|
+
"hosted_api_token_savings_claim_allowed": False,
|
|
2464
|
+
"hosted_api_cost_savings_claim_allowed": False,
|
|
2465
|
+
},
|
|
2466
|
+
"provenance_review": {
|
|
2467
|
+
"required": True,
|
|
2468
|
+
"reviewed": bool(missing_provenance_notes),
|
|
2469
|
+
"missing_provenance_notes": missing_provenance_notes,
|
|
2470
|
+
"missing_provenance_warnings": [] if missing_provenance_notes else ["missing_provenance_review"],
|
|
2471
|
+
"checkpoint_cannot_replace_raw_context_without_complete_provenance": True,
|
|
2472
|
+
},
|
|
2473
|
+
"review_plan": {
|
|
2474
|
+
"readiness_blockers": blockers,
|
|
2475
|
+
"missed_context_notes": missed_context_notes,
|
|
2476
|
+
"next_steps": [
|
|
2477
|
+
"Keep exact raw context fallback verified before checkpoint metadata is used.",
|
|
2478
|
+
"Deny protected evidence zones before any semantic checkpoint summary is considered.",
|
|
2479
|
+
"Keep provenance handles and missing-provenance review notes attached to checkpoint metadata.",
|
|
2480
|
+
"Measure provider/model token and cost fields on matched successful tasks before any hosted savings claim.",
|
|
2481
|
+
],
|
|
2482
|
+
},
|
|
2483
|
+
"claim_boundary": (
|
|
2484
|
+
"Dry-run semantic-checkpoint planning only; checkpoint metadata is not replacement context and no hosted "
|
|
2485
|
+
"token/cost savings claim is allowed without provider-measured matched successful tasks."
|
|
2486
|
+
),
|
|
2487
|
+
"candidate_replacement": None,
|
|
2488
|
+
}
|
|
2489
|
+
|
|
2490
|
+
|
|
2491
|
+
def command_plan_semantic_checkpoint(args: argparse.Namespace) -> int:
|
|
2492
|
+
payload = semantic_checkpoint_plan_payload(args)
|
|
2493
|
+
if args.json:
|
|
2494
|
+
emit_json(payload)
|
|
2495
|
+
else:
|
|
2496
|
+
print("ContextGuard semantic-checkpoint plan (dry-run only)")
|
|
2497
|
+
print("No files, prompts, transcripts, model/provider calls, proxy forwarding, or replacement context were emitted.")
|
|
2498
|
+
print(f"Status: {payload['status']}")
|
|
2499
|
+
if payload["review_plan"]["readiness_blockers"]:
|
|
2500
|
+
print(f"Readiness blockers: {', '.join(payload['review_plan']['readiness_blockers'])}")
|
|
2501
|
+
print(payload["claim_boundary"])
|
|
2502
|
+
return 0
|
|
2503
|
+
|
|
2504
|
+
|
|
2505
|
+
_PROOF_NONFINITE_SENTINEL = object()
|
|
2506
|
+
|
|
2507
|
+
|
|
2508
|
+
def ordered_proof_taxonomy(values: list[str] | set[str], order: tuple[str, ...]) -> list[str]:
|
|
2509
|
+
selected = set(values)
|
|
2510
|
+
return [value for value in order if value in selected]
|
|
2511
|
+
|
|
2512
|
+
|
|
2513
|
+
def empty_proof_unit_row(unit_index: int, issue: str) -> dict[str, Any]:
|
|
2514
|
+
return {
|
|
2515
|
+
"captured_at": None,
|
|
2516
|
+
"content_hash": {
|
|
2517
|
+
"algorithm": "sha256",
|
|
2518
|
+
"content_verified": False,
|
|
2519
|
+
"syntax_valid": False,
|
|
2520
|
+
"value": None,
|
|
2521
|
+
},
|
|
2522
|
+
"receipt": {
|
|
2523
|
+
"id": None,
|
|
2524
|
+
"storage_checked": False,
|
|
2525
|
+
"syntax_valid": False,
|
|
2526
|
+
},
|
|
2527
|
+
"rehydration": {
|
|
2528
|
+
"command": None,
|
|
2529
|
+
"executed": False,
|
|
2530
|
+
"receipt_bound": False,
|
|
2531
|
+
"syntax_valid": False,
|
|
2532
|
+
},
|
|
2533
|
+
"safe_range": None,
|
|
2534
|
+
"source_label": None,
|
|
2535
|
+
"syntax_and_consistency_valid": False,
|
|
2536
|
+
"transform_policy": None,
|
|
2537
|
+
"unit_index": unit_index,
|
|
2538
|
+
"validation_issues": [issue],
|
|
2539
|
+
"warnings": [],
|
|
2540
|
+
}
|
|
2541
|
+
|
|
2542
|
+
|
|
2543
|
+
def decode_proof_unit_json(raw: Any, unit_index: int) -> tuple[dict[str, Any] | None, dict[str, Any] | None]:
|
|
2544
|
+
if not isinstance(raw, str):
|
|
2545
|
+
return None, empty_proof_unit_row(unit_index, "invalid_proof_unit_json")
|
|
2546
|
+
try:
|
|
2547
|
+
encoded = raw.encode("utf-8", errors="strict")
|
|
2548
|
+
except UnicodeEncodeError:
|
|
2549
|
+
return None, empty_proof_unit_row(unit_index, "invalid_proof_unit_unicode")
|
|
2550
|
+
if len(encoded) > PROOF_CARRYING_CONTEXT_UNIT_JSON_BYTE_CAP:
|
|
2551
|
+
return None, empty_proof_unit_row(unit_index, "proof_unit_json_too_large")
|
|
2552
|
+
|
|
2553
|
+
duplicate_keys = False
|
|
2554
|
+
|
|
2555
|
+
def proof_object_pairs_hook(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
|
|
2556
|
+
nonlocal duplicate_keys
|
|
2557
|
+
result: dict[str, Any] = {}
|
|
2558
|
+
for key, value in pairs:
|
|
2559
|
+
if key in result:
|
|
2560
|
+
duplicate_keys = True
|
|
2561
|
+
result[key] = value
|
|
2562
|
+
return result
|
|
2563
|
+
|
|
2564
|
+
try:
|
|
2565
|
+
decoded = json.loads(
|
|
2566
|
+
raw,
|
|
2567
|
+
object_pairs_hook=proof_object_pairs_hook,
|
|
2568
|
+
parse_constant=lambda _value: _PROOF_NONFINITE_SENTINEL,
|
|
2569
|
+
)
|
|
2570
|
+
except RecursionError:
|
|
2571
|
+
return None, empty_proof_unit_row(unit_index, "proof_unit_json_nesting_too_deep")
|
|
2572
|
+
except (json.JSONDecodeError, ValueError, TypeError):
|
|
2573
|
+
return None, empty_proof_unit_row(unit_index, "invalid_proof_unit_json")
|
|
2574
|
+
|
|
2575
|
+
depth_exceeded = False
|
|
2576
|
+
decoded_unicode_invalid = False
|
|
2577
|
+
nonfinite_number = False
|
|
2578
|
+
stack: list[tuple[Any, int]] = [(decoded, 0)]
|
|
2579
|
+
while stack:
|
|
2580
|
+
value, depth = stack.pop()
|
|
2581
|
+
if depth > PROOF_UNIT_JSON_MAX_DEPTH:
|
|
2582
|
+
depth_exceeded = True
|
|
2583
|
+
if isinstance(value, str):
|
|
2584
|
+
try:
|
|
2585
|
+
value.encode("utf-8", errors="strict")
|
|
2586
|
+
except UnicodeEncodeError:
|
|
2587
|
+
decoded_unicode_invalid = True
|
|
2588
|
+
elif value is _PROOF_NONFINITE_SENTINEL:
|
|
2589
|
+
nonfinite_number = True
|
|
2590
|
+
elif type(value) is float and not math.isfinite(value):
|
|
2591
|
+
nonfinite_number = True
|
|
2592
|
+
|
|
2593
|
+
if depth > PROOF_UNIT_JSON_MAX_DEPTH:
|
|
2594
|
+
continue
|
|
2595
|
+
if isinstance(value, dict):
|
|
2596
|
+
next_depth = depth + 1
|
|
2597
|
+
for key, child in value.items():
|
|
2598
|
+
stack.append((key, next_depth))
|
|
2599
|
+
stack.append((child, next_depth))
|
|
2600
|
+
elif isinstance(value, list):
|
|
2601
|
+
next_depth = depth + 1
|
|
2602
|
+
for child in value:
|
|
2603
|
+
stack.append((child, next_depth))
|
|
2604
|
+
|
|
2605
|
+
if duplicate_keys:
|
|
2606
|
+
return None, empty_proof_unit_row(unit_index, "duplicate_proof_unit_keys")
|
|
2607
|
+
if depth_exceeded:
|
|
2608
|
+
return None, empty_proof_unit_row(unit_index, "proof_unit_json_nesting_too_deep")
|
|
2609
|
+
if nonfinite_number:
|
|
2610
|
+
return None, empty_proof_unit_row(unit_index, "nonfinite_proof_unit_number")
|
|
2611
|
+
if decoded_unicode_invalid:
|
|
2612
|
+
return None, empty_proof_unit_row(unit_index, "invalid_proof_unit_unicode")
|
|
2613
|
+
if not isinstance(decoded, dict):
|
|
2614
|
+
return None, empty_proof_unit_row(unit_index, "proof_unit_not_object")
|
|
2615
|
+
return decoded, None
|
|
2616
|
+
|
|
2617
|
+
|
|
2618
|
+
def normalize_required_proof_string(
|
|
2619
|
+
obj: dict[str, Any],
|
|
2620
|
+
field: str,
|
|
2621
|
+
missing_issue: str,
|
|
2622
|
+
invalid_issue: str,
|
|
2623
|
+
validator: Any,
|
|
2624
|
+
issues: list[str],
|
|
2625
|
+
) -> str | None:
|
|
2626
|
+
raw = obj.get(field)
|
|
2627
|
+
if raw is None or (isinstance(raw, str) and not raw.strip()):
|
|
2628
|
+
issues.append(missing_issue)
|
|
2629
|
+
return None
|
|
2630
|
+
if not isinstance(raw, str):
|
|
2631
|
+
issues.append(invalid_issue)
|
|
2632
|
+
return None
|
|
2633
|
+
value = raw.strip()
|
|
2634
|
+
if not validator(value):
|
|
2635
|
+
issues.append(invalid_issue)
|
|
2636
|
+
return None
|
|
2637
|
+
return value
|
|
2638
|
+
|
|
2639
|
+
|
|
2640
|
+
def valid_proof_timestamp(value: str) -> bool:
|
|
2641
|
+
if PROOF_CAPTURED_AT_RE.fullmatch(value) is None:
|
|
2642
|
+
return False
|
|
2643
|
+
try:
|
|
2644
|
+
datetime.strptime(value, "%Y-%m-%dT%H:%M:%SZ")
|
|
2645
|
+
except ValueError:
|
|
2646
|
+
return False
|
|
2647
|
+
return True
|
|
2648
|
+
|
|
2649
|
+
|
|
2650
|
+
def normalize_proof_safe_range(
|
|
2651
|
+
obj: dict[str, Any],
|
|
2652
|
+
issues: list[str],
|
|
2653
|
+
) -> tuple[dict[str, Any] | None, bool, bool]:
|
|
2654
|
+
if "safe_range" not in obj or obj.get("safe_range") is None:
|
|
2655
|
+
return None, False, False
|
|
2656
|
+
raw = obj.get("safe_range")
|
|
2657
|
+
if not isinstance(raw, dict) or set(raw) != {"kind", "start", "end"}:
|
|
2658
|
+
issues.append("invalid_safe_range")
|
|
2659
|
+
return None, True, True
|
|
2660
|
+
kind = raw.get("kind")
|
|
2661
|
+
start = raw.get("start")
|
|
2662
|
+
end = raw.get("end")
|
|
2663
|
+
if type(start) is not int or type(end) is not int:
|
|
2664
|
+
issues.append("invalid_safe_range")
|
|
2665
|
+
return None, True, True
|
|
2666
|
+
if kind == "lines":
|
|
2667
|
+
valid = 1 <= start <= end <= JSON_SAFE_INTEGER_MAX
|
|
2668
|
+
coordinate_system = "one_based_inclusive"
|
|
2669
|
+
elif kind == "bytes":
|
|
2670
|
+
valid = 0 <= start < end <= JSON_SAFE_INTEGER_MAX
|
|
2671
|
+
coordinate_system = "zero_based_half_open"
|
|
2672
|
+
else:
|
|
2673
|
+
valid = False
|
|
2674
|
+
coordinate_system = ""
|
|
2675
|
+
if not valid:
|
|
2676
|
+
issues.append("invalid_safe_range")
|
|
2677
|
+
return None, True, True
|
|
2678
|
+
return {
|
|
2679
|
+
"coordinate_system": coordinate_system,
|
|
2680
|
+
"end": end,
|
|
2681
|
+
"kind": kind,
|
|
2682
|
+
"start": start,
|
|
2683
|
+
}, True, False
|
|
2684
|
+
|
|
2685
|
+
|
|
2686
|
+
def parse_proof_rehydrate_command(command: str) -> tuple[bool, str | None]:
|
|
2687
|
+
if any(character in command for character in ";|&><`$\\\n\r"):
|
|
2688
|
+
return False, None
|
|
2689
|
+
try:
|
|
2690
|
+
argv = shlex.split(command)
|
|
2691
|
+
except ValueError:
|
|
2692
|
+
return False, None
|
|
2693
|
+
receipt: str | None = None
|
|
2694
|
+
if len(argv) == 4 and argv[0:2] == ["context-guard-artifact", "get"] and argv[3] == "--full":
|
|
2695
|
+
receipt = argv[2]
|
|
2696
|
+
elif len(argv) == 5 and argv[0:3] == ["context-guard", "artifact", "get"] and argv[4] == "--full":
|
|
2697
|
+
receipt = argv[3]
|
|
2698
|
+
if receipt is None or PROOF_RECEIPT_ID_RE.fullmatch(receipt) is None:
|
|
2699
|
+
return False, None
|
|
2700
|
+
return True, receipt
|
|
2701
|
+
|
|
2702
|
+
|
|
2703
|
+
def normalize_proof_unit(obj: dict[str, Any], unit_index: int) -> tuple[dict[str, Any], str | None, str | None]:
|
|
2704
|
+
issues: list[str] = []
|
|
2705
|
+
if set(obj) - PROOF_UNIT_ALLOWED_FIELDS:
|
|
2706
|
+
issues.append("unknown_proof_unit_fields")
|
|
2707
|
+
|
|
2708
|
+
source_label = normalize_required_proof_string(
|
|
2709
|
+
obj,
|
|
2710
|
+
"source_label",
|
|
2711
|
+
"missing_source_label",
|
|
2712
|
+
"invalid_source_label",
|
|
2713
|
+
lambda value: PROOF_SOURCE_LABEL_RE.fullmatch(value) is not None,
|
|
2714
|
+
issues,
|
|
2715
|
+
)
|
|
2716
|
+
receipt_id = normalize_required_proof_string(
|
|
2717
|
+
obj,
|
|
2718
|
+
"receipt_id",
|
|
2719
|
+
"missing_receipt",
|
|
2720
|
+
"invalid_receipt",
|
|
2721
|
+
lambda value: PROOF_RECEIPT_ID_RE.fullmatch(value) is not None,
|
|
2722
|
+
issues,
|
|
2723
|
+
)
|
|
2724
|
+
content_sha256 = normalize_required_proof_string(
|
|
2725
|
+
obj,
|
|
2726
|
+
"content_sha256",
|
|
2727
|
+
"missing_content_sha256",
|
|
2728
|
+
"invalid_content_sha256",
|
|
2729
|
+
lambda value: PROOF_CONTENT_SHA256_RE.fullmatch(value) is not None,
|
|
2730
|
+
issues,
|
|
2731
|
+
)
|
|
2732
|
+
captured_at = normalize_required_proof_string(
|
|
2733
|
+
obj,
|
|
2734
|
+
"captured_at",
|
|
2735
|
+
"missing_timestamp",
|
|
2736
|
+
"invalid_timestamp",
|
|
2737
|
+
valid_proof_timestamp,
|
|
2738
|
+
issues,
|
|
2739
|
+
)
|
|
2740
|
+
raw_transform_policy = obj.get("transform_policy")
|
|
2741
|
+
transform_policy: str | None = None
|
|
2742
|
+
if raw_transform_policy is None or (
|
|
2743
|
+
isinstance(raw_transform_policy, str) and not raw_transform_policy.strip()
|
|
2744
|
+
):
|
|
2745
|
+
issues.append("missing_transform_policy")
|
|
2746
|
+
elif not isinstance(raw_transform_policy, str) or raw_transform_policy not in {
|
|
2747
|
+
"identity",
|
|
2748
|
+
"safe_range_extract",
|
|
2749
|
+
}:
|
|
2750
|
+
issues.append("invalid_transform_policy")
|
|
2751
|
+
else:
|
|
2752
|
+
transform_policy = raw_transform_policy
|
|
2753
|
+
safe_range, safe_range_supplied, safe_range_invalid = normalize_proof_safe_range(obj, issues)
|
|
2754
|
+
if transform_policy == "safe_range_extract" and not safe_range_supplied and not safe_range_invalid:
|
|
2755
|
+
issues.append("missing_safe_range_for_transform_policy")
|
|
2756
|
+
|
|
2757
|
+
raw_command = obj.get("rehydrate_command")
|
|
2758
|
+
command_value: str | None = None
|
|
2759
|
+
command_syntax_valid = False
|
|
2760
|
+
command_receipt: str | None = None
|
|
2761
|
+
if raw_command is None or (isinstance(raw_command, str) and not raw_command.strip()):
|
|
2762
|
+
issues.append("missing_rehydrate_command")
|
|
2763
|
+
elif not isinstance(raw_command, str):
|
|
2764
|
+
issues.append("invalid_rehydrate_command")
|
|
2765
|
+
else:
|
|
2766
|
+
command_syntax_valid, command_receipt = parse_proof_rehydrate_command(raw_command)
|
|
2767
|
+
if not command_syntax_valid:
|
|
2768
|
+
issues.append("invalid_rehydrate_command")
|
|
2769
|
+
|
|
2770
|
+
receipt_bound = bool(
|
|
2771
|
+
command_syntax_valid
|
|
2772
|
+
and receipt_id is not None
|
|
2773
|
+
and command_receipt == receipt_id
|
|
2774
|
+
)
|
|
2775
|
+
if command_syntax_valid and receipt_id is not None and command_receipt != receipt_id:
|
|
2776
|
+
issues.append("rehydrate_receipt_mismatch")
|
|
2777
|
+
if receipt_bound and isinstance(raw_command, str):
|
|
2778
|
+
command_value = raw_command.strip()
|
|
2779
|
+
|
|
2780
|
+
ordered_issues = ordered_proof_taxonomy(issues, PROOF_READINESS_BLOCKER_ORDER)
|
|
2781
|
+
row = {
|
|
2782
|
+
"captured_at": captured_at,
|
|
2783
|
+
"content_hash": {
|
|
2784
|
+
"algorithm": "sha256",
|
|
2785
|
+
"content_verified": False,
|
|
2786
|
+
"syntax_valid": content_sha256 is not None,
|
|
2787
|
+
"value": content_sha256,
|
|
2788
|
+
},
|
|
2789
|
+
"receipt": {
|
|
2790
|
+
"id": receipt_id,
|
|
2791
|
+
"storage_checked": False,
|
|
2792
|
+
"syntax_valid": receipt_id is not None,
|
|
2793
|
+
},
|
|
2794
|
+
"rehydration": {
|
|
2795
|
+
"command": command_value,
|
|
2796
|
+
"executed": False,
|
|
2797
|
+
"receipt_bound": receipt_bound,
|
|
2798
|
+
"syntax_valid": command_syntax_valid,
|
|
2799
|
+
},
|
|
2800
|
+
"safe_range": safe_range,
|
|
2801
|
+
"source_label": source_label,
|
|
2802
|
+
"syntax_and_consistency_valid": not ordered_issues,
|
|
2803
|
+
"transform_policy": transform_policy,
|
|
2804
|
+
"unit_index": unit_index,
|
|
2805
|
+
"validation_issues": ordered_issues,
|
|
2806
|
+
"warnings": [],
|
|
2807
|
+
}
|
|
2808
|
+
return row, receipt_id, content_sha256
|
|
2809
|
+
|
|
2810
|
+
|
|
2811
|
+
def proof_duplicate_key(row: dict[str, Any]) -> tuple[Any, ...]:
|
|
2812
|
+
safe_range = row["safe_range"]
|
|
2813
|
+
normalized_range = None if safe_range is None else (
|
|
2814
|
+
safe_range["kind"],
|
|
2815
|
+
safe_range["start"],
|
|
2816
|
+
safe_range["end"],
|
|
2817
|
+
)
|
|
2818
|
+
return (
|
|
2819
|
+
row["source_label"],
|
|
2820
|
+
row["receipt"]["id"],
|
|
2821
|
+
row["content_hash"]["value"],
|
|
2822
|
+
normalized_range,
|
|
2823
|
+
row["captured_at"],
|
|
2824
|
+
row["transform_policy"],
|
|
2825
|
+
row["rehydration"]["command"],
|
|
2826
|
+
)
|
|
2827
|
+
|
|
2828
|
+
|
|
2829
|
+
def proof_verification_scope() -> dict[str, Any]:
|
|
2830
|
+
return {
|
|
2831
|
+
"content_hash_verified": False,
|
|
2832
|
+
"cross_field_consistency_checked": True,
|
|
2833
|
+
"cross_unit_receipt_hash_consistency_checked": True,
|
|
2834
|
+
"decoded_number_finiteness_checked": True,
|
|
2835
|
+
"decoded_unicode_checked": True,
|
|
2836
|
+
"duplicate_json_keys_checked": True,
|
|
2837
|
+
"field_syntax_checked": True,
|
|
2838
|
+
"json_depth_checked": True,
|
|
2839
|
+
"json_syntax_checked": True,
|
|
2840
|
+
"protected_zone_compliance_checked": False,
|
|
2841
|
+
"receipt_content_read": False,
|
|
2842
|
+
"receipt_storage_checked": False,
|
|
2843
|
+
"rehydration_executed": False,
|
|
2844
|
+
"safe_range_bounds_checked": False,
|
|
2845
|
+
"semantics": "validator_capability_invariant",
|
|
2846
|
+
"source_content_read": False,
|
|
2847
|
+
}
|
|
2848
|
+
|
|
2849
|
+
|
|
2850
|
+
def proof_carrying_context_plan_payload(args: argparse.Namespace) -> dict[str, Any]:
|
|
2851
|
+
raw_units = args.proof_unit_json or []
|
|
2852
|
+
supplied_count = len(raw_units)
|
|
2853
|
+
detailed_count = min(supplied_count, PROOF_CARRYING_CONTEXT_DETAILED_UNIT_CAP)
|
|
2854
|
+
overflow_count = max(supplied_count - PROOF_CARRYING_CONTEXT_DETAILED_UNIT_CAP, 0)
|
|
2855
|
+
detailed_raw_units = raw_units[:PROOF_CARRYING_CONTEXT_DETAILED_UNIT_CAP]
|
|
2856
|
+
|
|
2857
|
+
rows: list[dict[str, Any]] = []
|
|
2858
|
+
conflict_inputs: list[tuple[str | None, str | None]] = []
|
|
2859
|
+
for unit_index, raw in enumerate(detailed_raw_units):
|
|
2860
|
+
decoded, terminal_row = decode_proof_unit_json(raw, unit_index)
|
|
2861
|
+
if terminal_row is not None:
|
|
2862
|
+
rows.append(terminal_row)
|
|
2863
|
+
conflict_inputs.append((None, None))
|
|
2864
|
+
continue
|
|
2865
|
+
assert decoded is not None
|
|
2866
|
+
row, receipt_id, content_sha256 = normalize_proof_unit(decoded, unit_index)
|
|
2867
|
+
rows.append(row)
|
|
2868
|
+
conflict_inputs.append((receipt_id, content_sha256))
|
|
2869
|
+
|
|
2870
|
+
hashes_by_receipt: dict[str, set[str]] = {}
|
|
2871
|
+
for receipt_id, content_sha256 in conflict_inputs:
|
|
2872
|
+
if receipt_id is not None and content_sha256 is not None:
|
|
2873
|
+
hashes_by_receipt.setdefault(receipt_id, set()).add(content_sha256)
|
|
2874
|
+
conflicted_receipts = {
|
|
2875
|
+
receipt_id for receipt_id, hashes in hashes_by_receipt.items() if len(hashes) > 1
|
|
2876
|
+
}
|
|
2877
|
+
for row, (receipt_id, _content_sha256) in zip(rows, conflict_inputs):
|
|
2878
|
+
if receipt_id in conflicted_receipts:
|
|
2879
|
+
row["validation_issues"] = ordered_proof_taxonomy(
|
|
2880
|
+
[*row["validation_issues"], "receipt_hash_conflict"],
|
|
2881
|
+
PROOF_READINESS_BLOCKER_ORDER,
|
|
2882
|
+
)
|
|
2883
|
+
|
|
2884
|
+
duplicate_groups: dict[tuple[Any, ...], list[int]] = {}
|
|
2885
|
+
for row in rows:
|
|
2886
|
+
if not row["validation_issues"]:
|
|
2887
|
+
duplicate_groups.setdefault(proof_duplicate_key(row), []).append(row["unit_index"])
|
|
2888
|
+
duplicate_indexes = {
|
|
2889
|
+
unit_index
|
|
2890
|
+
for indexes in duplicate_groups.values()
|
|
2891
|
+
if len(indexes) > 1
|
|
2892
|
+
for unit_index in indexes
|
|
2893
|
+
}
|
|
2894
|
+
|
|
2895
|
+
valid_count = 0
|
|
2896
|
+
for row in rows:
|
|
2897
|
+
row_valid = not row["validation_issues"]
|
|
2898
|
+
row["syntax_and_consistency_valid"] = row_valid
|
|
2899
|
+
if not row_valid:
|
|
2900
|
+
row["warnings"] = []
|
|
2901
|
+
continue
|
|
2902
|
+
valid_count += 1
|
|
2903
|
+
warnings = list(PROOF_WARNING_ORDER[:6])
|
|
2904
|
+
if row["safe_range"] is None:
|
|
2905
|
+
warnings.append("safe_range_omitted")
|
|
2906
|
+
if row["unit_index"] in duplicate_indexes:
|
|
2907
|
+
warnings.append("duplicate_proof_unit")
|
|
2908
|
+
row["warnings"] = ordered_proof_taxonomy(warnings, PROOF_WARNING_ORDER)
|
|
2909
|
+
|
|
2910
|
+
protected_policy = (args.protected_zone_policy or "deny").strip().lower()
|
|
2911
|
+
blockers: list[str] = []
|
|
2912
|
+
if supplied_count == 0:
|
|
2913
|
+
blockers.append("missing_proof_unit")
|
|
2914
|
+
if overflow_count:
|
|
2915
|
+
blockers.append("too_many_proof_units")
|
|
2916
|
+
for row in rows:
|
|
2917
|
+
blockers.extend(row["validation_issues"])
|
|
2918
|
+
if protected_policy != "deny":
|
|
2919
|
+
blockers.append("protected_zone_denial_required")
|
|
2920
|
+
if not args.provider_boundary_ack:
|
|
2921
|
+
blockers.append("missing_provider_measurement_boundary")
|
|
2922
|
+
blockers = ordered_proof_taxonomy(blockers, PROOF_READINESS_BLOCKER_ORDER)
|
|
2923
|
+
|
|
2924
|
+
top_warnings = list(PROOF_WARNING_ORDER[:2])
|
|
2925
|
+
for row in rows:
|
|
2926
|
+
top_warnings.extend(row["warnings"])
|
|
2927
|
+
top_warnings = ordered_proof_taxonomy(top_warnings, PROOF_WARNING_ORDER)
|
|
2928
|
+
ready = not blockers
|
|
2929
|
+
|
|
2930
|
+
return {
|
|
2931
|
+
"candidate_replacement": None,
|
|
2932
|
+
"claim_boundary": (
|
|
2933
|
+
"Dry-run proof-carrying-context metadata validation only; protected-zone compliance, safe-range bounds, "
|
|
2934
|
+
"receipt storage, source content, SHA-256, timestamp freshness, and rehydration were not checked, no "
|
|
2935
|
+
"context was generated or replaced, and no hosted token/cost savings claim is allowed without "
|
|
2936
|
+
"provider-measured matched successful tasks."
|
|
2937
|
+
),
|
|
2938
|
+
"experiment_id": "proof-carrying-context",
|
|
2939
|
+
"external_services": {
|
|
2940
|
+
"called": False,
|
|
2941
|
+
"dns_lookup": False,
|
|
2942
|
+
"model_calls": False,
|
|
2943
|
+
"network": False,
|
|
2944
|
+
"provider_calls": False,
|
|
2945
|
+
"proxy_forwarding": False,
|
|
2946
|
+
},
|
|
2947
|
+
"measurement_boundary": {
|
|
2948
|
+
"hosted_api_cost_savings_claim_allowed": False,
|
|
2949
|
+
"hosted_api_token_savings_claim_allowed": False,
|
|
2950
|
+
"local_metadata_readiness_is_not_hosted_savings_evidence": True,
|
|
2951
|
+
"provider_boundary_acknowledged": bool(args.provider_boundary_ack),
|
|
2952
|
+
"provider_boundary_policy": IMAGE_CONTEXT_PACK_PROVIDER_BOUNDARY,
|
|
2953
|
+
"provider_measured_matched_successful_tasks_required_for_hosted_claims": True,
|
|
2954
|
+
"provider_model_specific": True,
|
|
2955
|
+
},
|
|
2956
|
+
"mode": "dry_run",
|
|
2957
|
+
"plan_only": {
|
|
2958
|
+
"command_advertised": True,
|
|
2959
|
+
"compact_context_generated": False,
|
|
2960
|
+
"emit_command_available": False,
|
|
2961
|
+
"evaluation_only": True,
|
|
2962
|
+
"record_command_available": False,
|
|
2963
|
+
"replacement_context_emitted": False,
|
|
2964
|
+
"runtime_behavior_changed": False,
|
|
2965
|
+
"serve_command_available": False,
|
|
2966
|
+
},
|
|
2967
|
+
"plan_schema_version": PROOF_CARRYING_CONTEXT_PLAN_SCHEMA_VERSION,
|
|
2968
|
+
"proof_contract": {
|
|
2969
|
+
"detailed_unit_cap": PROOF_CARRYING_CONTEXT_DETAILED_UNIT_CAP,
|
|
2970
|
+
"hash_policy": {
|
|
2971
|
+
"algorithm": "sha256",
|
|
2972
|
+
"content_read": False,
|
|
2973
|
+
"content_verified": False,
|
|
2974
|
+
"input_format": "64_lowercase_hex",
|
|
2975
|
+
},
|
|
2976
|
+
"optional_input_fields": ["safe_range"],
|
|
2977
|
+
"overflow_policy": {
|
|
2978
|
+
"detailed_rows_emitted": False,
|
|
2979
|
+
"overflow_values_echoed": False,
|
|
2980
|
+
"overflow_values_encoded": False,
|
|
2981
|
+
"overflow_values_parsed": False,
|
|
2982
|
+
},
|
|
2983
|
+
"proof_unit_input_flag": "--proof-unit-json",
|
|
2984
|
+
"proof_unit_input_repeatable": True,
|
|
2985
|
+
"rehydration_policy": {
|
|
2986
|
+
"allowed_command_shapes": [
|
|
2987
|
+
"context-guard-artifact get RECEIPT --full",
|
|
2988
|
+
"context-guard artifact get RECEIPT --full",
|
|
2989
|
+
],
|
|
2990
|
+
"command_executed": False,
|
|
2991
|
+
"receipt_bound_command_required": True,
|
|
2992
|
+
"receipt_storage_checked": False,
|
|
2993
|
+
},
|
|
2994
|
+
"required_input_fields": [
|
|
2995
|
+
"source_label",
|
|
2996
|
+
"receipt_id",
|
|
2997
|
+
"content_sha256",
|
|
2998
|
+
"captured_at",
|
|
2999
|
+
"transform_policy",
|
|
3000
|
+
"rehydrate_command",
|
|
3001
|
+
],
|
|
3002
|
+
"safe_range_policy": {
|
|
3003
|
+
"bounds_checked": False,
|
|
3004
|
+
"byte_coordinate_system": "zero_based_half_open",
|
|
3005
|
+
"json_safe_integer_max": JSON_SAFE_INTEGER_MAX,
|
|
3006
|
+
"kinds": ["lines", "bytes"],
|
|
3007
|
+
"line_coordinate_system": "one_based_inclusive",
|
|
3008
|
+
"required_by_default": False,
|
|
3009
|
+
"semantic_safety_checked": False,
|
|
3010
|
+
},
|
|
3011
|
+
"source_label_policy": {
|
|
3012
|
+
"max_characters": 120,
|
|
3013
|
+
"profile": "ascii-identifier-v1",
|
|
3014
|
+
"raw_content_allowed": False,
|
|
3015
|
+
"regex": "^[A-Za-z0-9][A-Za-z0-9._:/+-]{0,119}$",
|
|
3016
|
+
"safety_checked": False,
|
|
3017
|
+
"secrecy_checked": False,
|
|
3018
|
+
},
|
|
3019
|
+
"strict_json_policy": {
|
|
3020
|
+
"decoded_unicode_must_encode_utf8": True,
|
|
3021
|
+
"depth_root": 0,
|
|
3022
|
+
"duplicate_keys_allowed": False,
|
|
3023
|
+
"float_finiteness_check": "math.isfinite",
|
|
3024
|
+
"max_depth": PROOF_UNIT_JSON_MAX_DEPTH,
|
|
3025
|
+
"non_finite_numbers_allowed": False,
|
|
3026
|
+
"post_decode_walk": "iterative_container_keys_and_values",
|
|
3027
|
+
"raw_unicode_must_encode_utf8": True,
|
|
3028
|
+
"root_must_be_object": True,
|
|
3029
|
+
},
|
|
3030
|
+
"timestamp_policy": {
|
|
3031
|
+
"caller_supplied_only": True,
|
|
3032
|
+
"current_time_generated": False,
|
|
3033
|
+
"freshness_checked": False,
|
|
3034
|
+
"input_format": "YYYY-MM-DDTHH:MM:SSZ",
|
|
3035
|
+
"required": True,
|
|
3036
|
+
},
|
|
3037
|
+
"transform_policy": {
|
|
3038
|
+
"allowed": ["identity", "safe_range_extract"],
|
|
3039
|
+
"automatic_deletion_allowed": False,
|
|
3040
|
+
"lossy_transform_allowed": False,
|
|
3041
|
+
"semantic_rewrite_allowed": False,
|
|
3042
|
+
},
|
|
3043
|
+
"unit_json_byte_cap": PROOF_CARRYING_CONTEXT_UNIT_JSON_BYTE_CAP,
|
|
3044
|
+
"verification_scope": proof_verification_scope(),
|
|
3045
|
+
},
|
|
3046
|
+
"proof_unit_schema_version": PROOF_CARRYING_CONTEXT_UNIT_SCHEMA_VERSION,
|
|
3047
|
+
"proof_units": rows,
|
|
3048
|
+
"protected_zones": {
|
|
3049
|
+
"compliance_checked": False,
|
|
3050
|
+
"content_inspected": False,
|
|
3051
|
+
"declared_policy": protected_policy,
|
|
3052
|
+
"declared_policy_only": True,
|
|
3053
|
+
"denied_classes": [
|
|
3054
|
+
"code",
|
|
3055
|
+
"diffs",
|
|
3056
|
+
"identifiers",
|
|
3057
|
+
"hashes",
|
|
3058
|
+
"paths",
|
|
3059
|
+
"numeric_constants",
|
|
3060
|
+
"json_keys",
|
|
3061
|
+
"stack_frames",
|
|
3062
|
+
"secrets",
|
|
3063
|
+
"prompt_like_instructions",
|
|
3064
|
+
],
|
|
3065
|
+
"override_allowed": False,
|
|
3066
|
+
"prompt_like_instruction_compliance_checked": False,
|
|
3067
|
+
"semantic_transform_permitted_by_gate": False,
|
|
3068
|
+
},
|
|
3069
|
+
"review_plan": {
|
|
3070
|
+
"detailed_proof_unit_count": detailed_count,
|
|
3071
|
+
"invalid_detailed_proof_unit_count": detailed_count - valid_count,
|
|
3072
|
+
"next_steps": [
|
|
3073
|
+
"Treat this result as proof-envelope metadata syntax and consistency review only.",
|
|
3074
|
+
"Verify protected-zone compliance, range bounds, receipt storage, source content, SHA-256, and rehydration only in a separately approved future consumer/runtime.",
|
|
3075
|
+
"Keep protected evidence and prompt-like instructions out of transformation paths.",
|
|
3076
|
+
"Measure provider/model token and cost fields on matched successful tasks before any hosted savings claim.",
|
|
3077
|
+
],
|
|
3078
|
+
"overflow_proof_unit_count": overflow_count,
|
|
3079
|
+
"readiness_blocker_order": list(PROOF_READINESS_BLOCKER_ORDER),
|
|
3080
|
+
"readiness_blockers": blockers,
|
|
3081
|
+
"supplied_proof_unit_count": supplied_count,
|
|
3082
|
+
"valid_detailed_proof_unit_count": valid_count,
|
|
3083
|
+
"warning_order": list(PROOF_WARNING_ORDER),
|
|
3084
|
+
"warnings": top_warnings,
|
|
3085
|
+
},
|
|
3086
|
+
"runtime_side_effects": {
|
|
3087
|
+
"artifact_files_read": False,
|
|
3088
|
+
"config_files_read": False,
|
|
3089
|
+
"current_time_generated": False,
|
|
3090
|
+
"files_written": False,
|
|
3091
|
+
"prompt_edited": False,
|
|
3092
|
+
"rehydrate_command_executed": False,
|
|
3093
|
+
"source_files_read": False,
|
|
3094
|
+
"stable_runtime_behavior_changed": False,
|
|
3095
|
+
"stdin_content_read": False,
|
|
3096
|
+
"subprocesses_executed": False,
|
|
3097
|
+
"transcript_edited": False,
|
|
3098
|
+
},
|
|
3099
|
+
"schema_version": CONFIG_SCHEMA_VERSION,
|
|
3100
|
+
"status": (
|
|
3101
|
+
"ready_for_plan_review"
|
|
3102
|
+
if ready
|
|
3103
|
+
else "blocked_until_proof_carrying_context_gate_ready"
|
|
3104
|
+
),
|
|
3105
|
+
"tool": TOOL_NAME,
|
|
3106
|
+
}
|
|
3107
|
+
|
|
3108
|
+
|
|
3109
|
+
def command_plan_proof_carrying_context(args: argparse.Namespace) -> int:
|
|
3110
|
+
payload = proof_carrying_context_plan_payload(args)
|
|
3111
|
+
if args.json:
|
|
3112
|
+
emit_json(payload)
|
|
3113
|
+
else:
|
|
3114
|
+
print("ContextGuard proof-carrying-context plan (dry-run metadata readiness only)")
|
|
3115
|
+
print("No source/artifact/config/stdin content was read; no verification, context generation, replacement, network, subprocess, or file write occurred.")
|
|
3116
|
+
print(f"Status: {payload['status']}")
|
|
3117
|
+
if payload["review_plan"]["readiness_blockers"]:
|
|
3118
|
+
print(f"Readiness blockers: {', '.join(payload['review_plan']['readiness_blockers'])}")
|
|
3119
|
+
print(f"Warnings: {', '.join(payload['review_plan']['warnings'])}")
|
|
3120
|
+
print(payload["claim_boundary"])
|
|
3121
|
+
return 0
|
|
3122
|
+
|
|
3123
|
+
|
|
3124
|
+
def proof_verification_row(unit_index: int, blockers: list[str] | None = None) -> dict[str, Any]:
|
|
3125
|
+
return {
|
|
3126
|
+
"blockers": list(blockers or []),
|
|
3127
|
+
"content_hash": {
|
|
3128
|
+
"algorithm": "sha256",
|
|
3129
|
+
"declared_value": None,
|
|
3130
|
+
"matches_proof_unit": False,
|
|
3131
|
+
"matches_receipt_metadata": False,
|
|
3132
|
+
"verified": False,
|
|
3133
|
+
},
|
|
3134
|
+
"preflight_valid": False,
|
|
3135
|
+
"protected_zone": {"compliance_checked": False, "status": "unchecked"},
|
|
3136
|
+
"receipt": {
|
|
3137
|
+
"content_file_verified": False,
|
|
3138
|
+
"id": None,
|
|
3139
|
+
"metadata_file_verified": False,
|
|
3140
|
+
"metadata_verified": False,
|
|
3141
|
+
"stored_bytes": None,
|
|
3142
|
+
"stored_lines": None,
|
|
3143
|
+
"verified": False,
|
|
3144
|
+
},
|
|
3145
|
+
"rehydration": {
|
|
3146
|
+
"executed": False,
|
|
3147
|
+
"receipt_bound": False,
|
|
3148
|
+
"syntax_valid": False,
|
|
3149
|
+
"verified": False,
|
|
3150
|
+
},
|
|
3151
|
+
"safe_range": None,
|
|
3152
|
+
"source_label": None,
|
|
3153
|
+
"status": "verification_failed",
|
|
3154
|
+
"timestamp": {
|
|
3155
|
+
"format_valid": False,
|
|
3156
|
+
"freshness_checked": False,
|
|
3157
|
+
"status": "invalid_or_unavailable",
|
|
3158
|
+
},
|
|
3159
|
+
"transform_policy": None,
|
|
3160
|
+
"unit_index": unit_index,
|
|
3161
|
+
"warnings": [],
|
|
3162
|
+
}
|
|
3163
|
+
|
|
3164
|
+
|
|
3165
|
+
def normalized_proof_verification_row(plan_row: dict[str, Any]) -> dict[str, Any]:
|
|
3166
|
+
row = proof_verification_row(plan_row["unit_index"])
|
|
3167
|
+
safe_range = plan_row["safe_range"]
|
|
3168
|
+
row.update({
|
|
3169
|
+
"preflight_valid": True,
|
|
3170
|
+
"source_label": plan_row["source_label"],
|
|
3171
|
+
"transform_policy": plan_row["transform_policy"],
|
|
3172
|
+
"timestamp": {
|
|
3173
|
+
"format_valid": True,
|
|
3174
|
+
"freshness_checked": False,
|
|
3175
|
+
"status": "format_valid_semantics_unchecked",
|
|
3176
|
+
},
|
|
3177
|
+
})
|
|
3178
|
+
row["receipt"]["id"] = plan_row["receipt"]["id"]
|
|
3179
|
+
row["content_hash"]["declared_value"] = plan_row["content_hash"]["value"]
|
|
3180
|
+
row["rehydration"] = {
|
|
3181
|
+
"executed": False,
|
|
3182
|
+
"receipt_bound": True,
|
|
3183
|
+
"syntax_valid": True,
|
|
3184
|
+
"verified": True,
|
|
3185
|
+
}
|
|
3186
|
+
if safe_range is not None:
|
|
3187
|
+
row["safe_range"] = {
|
|
3188
|
+
"bounds_checked": False,
|
|
3189
|
+
"coordinate_system": safe_range["coordinate_system"],
|
|
3190
|
+
"end": safe_range["end"],
|
|
3191
|
+
"kind": safe_range["kind"],
|
|
3192
|
+
"range_content_retrieved": False,
|
|
3193
|
+
"start": safe_range["start"],
|
|
3194
|
+
"status": "not_checked",
|
|
3195
|
+
}
|
|
3196
|
+
return row
|
|
3197
|
+
|
|
3198
|
+
|
|
3199
|
+
def proof_artifact_io_capabilities_available() -> bool:
|
|
3200
|
+
return bool(
|
|
3201
|
+
NO_FOLLOW_SUPPORTED
|
|
3202
|
+
and hasattr(os, "O_NOFOLLOW")
|
|
3203
|
+
and DIR_FD_OPEN_SUPPORTED
|
|
3204
|
+
and DIR_FD_STAT_NOFOLLOW_SUPPORTED
|
|
3205
|
+
and callable(getattr(os, "open", None))
|
|
3206
|
+
and callable(getattr(os, "stat", None))
|
|
3207
|
+
and callable(getattr(os, "fstat", None))
|
|
3208
|
+
and callable(getattr(os, "geteuid", None))
|
|
3209
|
+
and callable(getattr(os, "read", None))
|
|
3210
|
+
and callable(getattr(os, "close", None))
|
|
3211
|
+
)
|
|
3212
|
+
|
|
3213
|
+
|
|
3214
|
+
def validate_proof_artifact_dir_arg(raw: Any) -> tuple[str | None, str | None]:
|
|
3215
|
+
if not isinstance(raw, str) or not raw or "\x00" in raw:
|
|
3216
|
+
return None, "invalid_artifact_directory"
|
|
3217
|
+
components = raw.split("/")
|
|
3218
|
+
if ".." in components:
|
|
3219
|
+
return None, "invalid_artifact_directory"
|
|
3220
|
+
if not raw.startswith("/"):
|
|
3221
|
+
first = next((component for component in components if component), "")
|
|
3222
|
+
if first == "~" or first.startswith("~"):
|
|
3223
|
+
return None, "invalid_artifact_directory"
|
|
3224
|
+
normalized = os.path.normpath(raw)
|
|
3225
|
+
if normalized.startswith("//"):
|
|
3226
|
+
normalized = "/" + normalized.lstrip("/")
|
|
3227
|
+
if not os.path.isabs(normalized):
|
|
3228
|
+
normalized = os.path.normpath(os.path.join(os.getcwd(), normalized))
|
|
3229
|
+
return normalized, None
|
|
3230
|
+
|
|
3231
|
+
|
|
3232
|
+
def normalize_proof_allowed_macos_alias(path: str) -> str:
|
|
3233
|
+
parts = Path(path).parts
|
|
3234
|
+
if len(parts) < 2 or parts[1] not in {"tmp", "var"}:
|
|
3235
|
+
return path
|
|
3236
|
+
alias = "/" + parts[1]
|
|
3237
|
+
expected = "/private/" + parts[1]
|
|
3238
|
+
try:
|
|
3239
|
+
info = os.stat(alias, follow_symlinks=False)
|
|
3240
|
+
if (
|
|
3241
|
+
stat.S_ISLNK(info.st_mode)
|
|
3242
|
+
and str(_normalized_link_target(Path("/"), os.readlink(alias))) == expected
|
|
3243
|
+
):
|
|
3244
|
+
return os.path.join(expected, *parts[2:])
|
|
3245
|
+
except OSError:
|
|
3246
|
+
pass
|
|
3247
|
+
return path
|
|
3248
|
+
|
|
3249
|
+
|
|
3250
|
+
def map_proof_directory_error(exc: OSError, parent_fd: int | None, component: str | None) -> str:
|
|
3251
|
+
if exc.errno == errno.ENOENT:
|
|
3252
|
+
return "artifact_directory_not_found"
|
|
3253
|
+
if exc.errno == errno.ELOOP:
|
|
3254
|
+
return "artifact_directory_symlink_rejected"
|
|
3255
|
+
if exc.errno == errno.ENOTDIR:
|
|
3256
|
+
if parent_fd is not None and component is not None:
|
|
3257
|
+
try:
|
|
3258
|
+
info = os.stat(component, dir_fd=parent_fd, follow_symlinks=False)
|
|
3259
|
+
if stat.S_ISLNK(info.st_mode):
|
|
3260
|
+
return "artifact_directory_symlink_rejected"
|
|
3261
|
+
except OSError:
|
|
3262
|
+
pass
|
|
3263
|
+
return "artifact_directory_not_regular"
|
|
3264
|
+
if exc.errno in {errno.EACCES, errno.EPERM}:
|
|
3265
|
+
return "artifact_directory_access_failed"
|
|
3266
|
+
return "artifact_directory_access_failed"
|
|
3267
|
+
|
|
3268
|
+
|
|
3269
|
+
def open_proof_artifact_directory(path: str) -> tuple[int | None, str | None]:
|
|
3270
|
+
path = normalize_proof_allowed_macos_alias(path)
|
|
3271
|
+
flags = os.O_RDONLY | os.O_NOFOLLOW
|
|
3272
|
+
if hasattr(os, "O_CLOEXEC"):
|
|
3273
|
+
flags |= os.O_CLOEXEC
|
|
3274
|
+
if hasattr(os, "O_DIRECTORY"):
|
|
3275
|
+
flags |= os.O_DIRECTORY
|
|
3276
|
+
current_fd: int | None = None
|
|
3277
|
+
try:
|
|
3278
|
+
current_fd = os.open("/", flags)
|
|
3279
|
+
for component in (part for part in Path(path).parts[1:] if part not in {"", "."}):
|
|
3280
|
+
try:
|
|
3281
|
+
next_fd = os.open(component, flags, dir_fd=current_fd)
|
|
3282
|
+
except OSError as exc:
|
|
3283
|
+
return None, map_proof_directory_error(exc, current_fd, component)
|
|
3284
|
+
os.close(current_fd)
|
|
3285
|
+
current_fd = next_fd
|
|
3286
|
+
try:
|
|
3287
|
+
info = os.fstat(current_fd)
|
|
3288
|
+
except OSError:
|
|
3289
|
+
return None, "artifact_directory_access_failed"
|
|
3290
|
+
if not stat.S_ISDIR(info.st_mode):
|
|
3291
|
+
return None, "artifact_directory_not_regular"
|
|
3292
|
+
if info.st_uid != os.geteuid():
|
|
3293
|
+
return None, "artifact_directory_owner_mismatch"
|
|
3294
|
+
if stat.S_IMODE(info.st_mode) != 0o700:
|
|
3295
|
+
return None, "artifact_directory_mode_not_private"
|
|
3296
|
+
result = current_fd
|
|
3297
|
+
current_fd = None
|
|
3298
|
+
return result, None
|
|
3299
|
+
except OSError as exc:
|
|
3300
|
+
return None, map_proof_directory_error(exc, None, None)
|
|
3301
|
+
finally:
|
|
3302
|
+
if current_fd is not None:
|
|
3303
|
+
try:
|
|
3304
|
+
os.close(current_fd)
|
|
3305
|
+
except OSError:
|
|
3306
|
+
pass
|
|
3307
|
+
|
|
3308
|
+
|
|
3309
|
+
def proof_leaf_stat_blockers(info: Any, leaf_kind: str) -> list[str]:
|
|
3310
|
+
prefix = f"receipt_{leaf_kind}"
|
|
3311
|
+
if stat.S_ISLNK(info.st_mode):
|
|
3312
|
+
return [f"{prefix}_symlink_rejected"]
|
|
3313
|
+
if not stat.S_ISREG(info.st_mode):
|
|
3314
|
+
return [f"{prefix}_not_regular"]
|
|
3315
|
+
blockers: list[str] = []
|
|
3316
|
+
if info.st_uid != os.geteuid():
|
|
3317
|
+
blockers.append(f"{prefix}_owner_mismatch")
|
|
3318
|
+
if stat.S_IMODE(info.st_mode) != 0o600:
|
|
3319
|
+
blockers.append(f"{prefix}_mode_not_private")
|
|
3320
|
+
if info.st_nlink != 1:
|
|
3321
|
+
blockers.append(f"{prefix}_multiple_links")
|
|
3322
|
+
return blockers
|
|
3323
|
+
|
|
3324
|
+
|
|
3325
|
+
def proof_leaf_precheck(
|
|
3326
|
+
artifact_fd: int,
|
|
3327
|
+
name: str,
|
|
3328
|
+
leaf_kind: str,
|
|
3329
|
+
) -> tuple[Any | None, bool, list[str]]:
|
|
3330
|
+
try:
|
|
3331
|
+
info = os.stat(name, dir_fd=artifact_fd, follow_symlinks=False)
|
|
3332
|
+
except OSError as exc:
|
|
3333
|
+
if exc.errno == errno.ENOENT:
|
|
3334
|
+
return None, True, []
|
|
3335
|
+
if exc.errno == errno.ELOOP:
|
|
3336
|
+
return None, False, [f"receipt_{leaf_kind}_symlink_rejected"]
|
|
3337
|
+
return None, False, ["artifact_read_failed"]
|
|
3338
|
+
return info, False, proof_leaf_stat_blockers(info, leaf_kind)
|
|
3339
|
+
|
|
3340
|
+
|
|
3341
|
+
def proof_stat_stability(info: Any) -> tuple[int, ...]:
|
|
3342
|
+
"""Return every identity, mutation, and leaf-policy field used by verification."""
|
|
3343
|
+
return (
|
|
3344
|
+
info.st_dev,
|
|
3345
|
+
info.st_ino,
|
|
3346
|
+
info.st_size,
|
|
3347
|
+
info.st_mtime_ns,
|
|
3348
|
+
info.st_ctime_ns,
|
|
3349
|
+
info.st_uid,
|
|
3350
|
+
info.st_gid,
|
|
3351
|
+
stat.S_IFMT(info.st_mode),
|
|
3352
|
+
stat.S_IMODE(info.st_mode),
|
|
3353
|
+
info.st_nlink,
|
|
3354
|
+
)
|
|
3355
|
+
|
|
3356
|
+
|
|
3357
|
+
_PROOF_METADATA_NONFINITE_SENTINEL = object()
|
|
3358
|
+
|
|
3359
|
+
|
|
3360
|
+
def decode_proof_receipt_metadata(raw: bytes, receipt: str) -> tuple[dict[str, Any] | None, list[str]]:
|
|
3361
|
+
try:
|
|
3362
|
+
text = raw.decode("utf-8", errors="strict")
|
|
3363
|
+
except UnicodeDecodeError:
|
|
3364
|
+
return None, ["receipt_metadata_invalid_unicode"]
|
|
3365
|
+
duplicate_keys = False
|
|
3366
|
+
|
|
3367
|
+
def pairs_hook(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
|
|
3368
|
+
nonlocal duplicate_keys
|
|
3369
|
+
obj: dict[str, Any] = {}
|
|
3370
|
+
for key, value in pairs:
|
|
3371
|
+
if key in obj:
|
|
3372
|
+
duplicate_keys = True
|
|
3373
|
+
obj[key] = value
|
|
3374
|
+
return obj
|
|
3375
|
+
|
|
3376
|
+
try:
|
|
3377
|
+
decoded = json.loads(
|
|
3378
|
+
text,
|
|
3379
|
+
object_pairs_hook=pairs_hook,
|
|
3380
|
+
parse_constant=lambda _value: _PROOF_METADATA_NONFINITE_SENTINEL,
|
|
3381
|
+
)
|
|
3382
|
+
except RecursionError:
|
|
3383
|
+
return None, ["receipt_metadata_nesting_too_deep"]
|
|
3384
|
+
except (json.JSONDecodeError, ValueError, TypeError):
|
|
3385
|
+
return None, ["receipt_metadata_invalid_json"]
|
|
3386
|
+
|
|
3387
|
+
depth_exceeded = False
|
|
3388
|
+
decoded_unicode_invalid = False
|
|
3389
|
+
nonfinite = False
|
|
3390
|
+
stack: list[tuple[Any, int]] = [(decoded, 0)]
|
|
3391
|
+
while stack:
|
|
3392
|
+
value, depth = stack.pop()
|
|
3393
|
+
if depth > PROOF_UNIT_JSON_MAX_DEPTH:
|
|
3394
|
+
depth_exceeded = True
|
|
3395
|
+
continue
|
|
3396
|
+
if isinstance(value, str):
|
|
3397
|
+
try:
|
|
3398
|
+
value.encode("utf-8", errors="strict")
|
|
3399
|
+
except UnicodeEncodeError:
|
|
3400
|
+
decoded_unicode_invalid = True
|
|
3401
|
+
if value is _PROOF_METADATA_NONFINITE_SENTINEL or (
|
|
3402
|
+
type(value) is float and not math.isfinite(value)
|
|
3403
|
+
):
|
|
3404
|
+
nonfinite = True
|
|
3405
|
+
if isinstance(value, dict):
|
|
3406
|
+
for key, child in value.items():
|
|
3407
|
+
stack.append((key, depth + 1))
|
|
3408
|
+
stack.append((child, depth + 1))
|
|
3409
|
+
elif isinstance(value, list):
|
|
3410
|
+
for child in value:
|
|
3411
|
+
stack.append((child, depth + 1))
|
|
3412
|
+
if duplicate_keys:
|
|
3413
|
+
return None, ["receipt_metadata_duplicate_keys"]
|
|
3414
|
+
if depth_exceeded:
|
|
3415
|
+
return None, ["receipt_metadata_nesting_too_deep"]
|
|
3416
|
+
if nonfinite:
|
|
3417
|
+
return None, ["receipt_metadata_nonfinite_number"]
|
|
3418
|
+
if decoded_unicode_invalid:
|
|
3419
|
+
return None, ["receipt_metadata_invalid_unicode"]
|
|
3420
|
+
if not isinstance(decoded, dict):
|
|
3421
|
+
return None, ["receipt_metadata_not_object"]
|
|
3422
|
+
if decoded.get("artifact_id") != receipt:
|
|
3423
|
+
return None, ["receipt_metadata_id_mismatch"]
|
|
3424
|
+
stored = decoded.get("stored_output")
|
|
3425
|
+
if not isinstance(stored, dict):
|
|
3426
|
+
return None, ["receipt_metadata_stored_output_invalid"]
|
|
3427
|
+
sha256 = stored.get("sha256")
|
|
3428
|
+
byte_count = stored.get("bytes")
|
|
3429
|
+
line_count = stored.get("lines")
|
|
3430
|
+
stored_valid = bool(
|
|
3431
|
+
isinstance(sha256, str)
|
|
3432
|
+
and PROOF_CONTENT_SHA256_RE.fullmatch(sha256)
|
|
3433
|
+
and type(byte_count) is int
|
|
3434
|
+
and 0 <= byte_count <= PROOF_RECEIPT_CONTENT_BYTE_CAP
|
|
3435
|
+
and type(line_count) is int
|
|
3436
|
+
and (
|
|
3437
|
+
(byte_count == 0 and line_count == 0)
|
|
3438
|
+
or (byte_count > 0 and 1 <= line_count <= byte_count)
|
|
3439
|
+
)
|
|
3440
|
+
)
|
|
3441
|
+
if not stored_valid:
|
|
3442
|
+
return None, ["receipt_metadata_stored_output_invalid"]
|
|
3443
|
+
if (
|
|
3444
|
+
stored.get("content_file") != f"{receipt}.txt"
|
|
3445
|
+
or stored.get("metadata_file") != f"{receipt}.json"
|
|
3446
|
+
):
|
|
3447
|
+
return None, ["receipt_metadata_file_binding_mismatch"]
|
|
3448
|
+
return decoded, []
|
|
3449
|
+
|
|
3450
|
+
|
|
3451
|
+
def open_proof_leaf(
|
|
3452
|
+
artifact_fd: int,
|
|
3453
|
+
name: str,
|
|
3454
|
+
precheck: Any,
|
|
3455
|
+
leaf_kind: str,
|
|
3456
|
+
) -> tuple[int | None, Any | None, list[str]]:
|
|
3457
|
+
try:
|
|
3458
|
+
fd = os.open(
|
|
3459
|
+
name,
|
|
3460
|
+
_file_open_flags(label=f"proof receipt {leaf_kind}"),
|
|
3461
|
+
dir_fd=artifact_fd,
|
|
3462
|
+
)
|
|
3463
|
+
except OSError as exc:
|
|
3464
|
+
if exc.errno == errno.ELOOP:
|
|
3465
|
+
return None, None, [f"receipt_{leaf_kind}_symlink_rejected"]
|
|
3466
|
+
return None, None, ["artifact_read_failed"]
|
|
3467
|
+
try:
|
|
3468
|
+
opened = os.fstat(fd)
|
|
3469
|
+
blockers = proof_leaf_stat_blockers(opened, leaf_kind)
|
|
3470
|
+
if proof_stat_stability(opened) != proof_stat_stability(precheck):
|
|
3471
|
+
blockers.append("artifact_changed_during_read")
|
|
3472
|
+
cap = (
|
|
3473
|
+
PROOF_RECEIPT_METADATA_BYTE_CAP
|
|
3474
|
+
if leaf_kind == "metadata"
|
|
3475
|
+
else PROOF_RECEIPT_CONTENT_BYTE_CAP
|
|
3476
|
+
)
|
|
3477
|
+
if opened.st_size > cap:
|
|
3478
|
+
blockers.append(f"receipt_{leaf_kind}_too_large")
|
|
3479
|
+
blockers = ordered_proof_taxonomy(blockers, PROOF_VERIFICATION_BLOCKER_ORDER)
|
|
3480
|
+
if blockers:
|
|
3481
|
+
os.close(fd)
|
|
3482
|
+
return None, None, blockers
|
|
3483
|
+
return fd, opened, []
|
|
3484
|
+
except OSError:
|
|
3485
|
+
try:
|
|
3486
|
+
os.close(fd)
|
|
3487
|
+
except OSError:
|
|
3488
|
+
pass
|
|
3489
|
+
return None, None, ["artifact_read_failed"]
|
|
3490
|
+
|
|
3491
|
+
|
|
3492
|
+
def read_proof_metadata_leaf(fd: int, opened: Any) -> tuple[bytes | None, list[str]]:
|
|
3493
|
+
if opened.st_size > PROOF_RECEIPT_METADATA_BYTE_CAP:
|
|
3494
|
+
return None, ["receipt_metadata_too_large"]
|
|
3495
|
+
expected_size = opened.st_size
|
|
3496
|
+
accumulated = bytearray()
|
|
3497
|
+
first_read = True
|
|
3498
|
+
eof = False
|
|
3499
|
+
try:
|
|
3500
|
+
while first_read or len(accumulated) != expected_size:
|
|
3501
|
+
first_read = False
|
|
3502
|
+
if len(accumulated) >= PROOF_RECEIPT_METADATA_BYTE_CAP + 1:
|
|
3503
|
+
break
|
|
3504
|
+
chunk = os.read(
|
|
3505
|
+
fd,
|
|
3506
|
+
PROOF_RECEIPT_METADATA_BYTE_CAP + 1 - len(accumulated),
|
|
3507
|
+
)
|
|
3508
|
+
if not chunk:
|
|
3509
|
+
eof = True
|
|
3510
|
+
break
|
|
3511
|
+
accumulated.extend(chunk)
|
|
3512
|
+
if len(accumulated) == expected_size:
|
|
3513
|
+
break
|
|
3514
|
+
except OSError:
|
|
3515
|
+
return None, ["artifact_read_failed"]
|
|
3516
|
+
blockers = []
|
|
3517
|
+
if len(accumulated) > PROOF_RECEIPT_METADATA_BYTE_CAP:
|
|
3518
|
+
blockers.append("receipt_metadata_too_large")
|
|
3519
|
+
if len(accumulated) != expected_size or (eof and len(accumulated) < expected_size):
|
|
3520
|
+
blockers.append("artifact_changed_during_read")
|
|
3521
|
+
blockers = ordered_proof_taxonomy(blockers, PROOF_VERIFICATION_BLOCKER_ORDER)
|
|
3522
|
+
return (None, blockers) if blockers else (bytes(accumulated), [])
|
|
3523
|
+
|
|
3524
|
+
|
|
3525
|
+
def read_proof_content_leaf(
|
|
3526
|
+
fd: int,
|
|
3527
|
+
opened: Any,
|
|
3528
|
+
declared_bytes: int,
|
|
3529
|
+
runtime_boundaries: dict[str, Any],
|
|
3530
|
+
) -> tuple[bytes | None, list[str]]:
|
|
3531
|
+
blockers = []
|
|
3532
|
+
if opened.st_size > PROOF_RECEIPT_CONTENT_BYTE_CAP:
|
|
3533
|
+
blockers.append("receipt_content_too_large")
|
|
3534
|
+
if opened.st_size != declared_bytes:
|
|
3535
|
+
blockers.append("receipt_content_size_mismatch")
|
|
3536
|
+
blockers = ordered_proof_taxonomy(blockers, PROOF_VERIFICATION_BLOCKER_ORDER)
|
|
3537
|
+
if blockers:
|
|
3538
|
+
return None, blockers
|
|
3539
|
+
accumulated = bytearray()
|
|
3540
|
+
early_eof = False
|
|
3541
|
+
extra_data = False
|
|
3542
|
+
try:
|
|
3543
|
+
while len(accumulated) < declared_bytes:
|
|
3544
|
+
runtime_boundaries["artifact_content_read_for_whole_file_verification"] = True
|
|
3545
|
+
chunk = os.read(
|
|
3546
|
+
fd,
|
|
3547
|
+
min(PROOF_RECEIPT_CONTENT_READ_CHUNK, declared_bytes - len(accumulated)),
|
|
3548
|
+
)
|
|
3549
|
+
if not chunk:
|
|
3550
|
+
early_eof = True
|
|
3551
|
+
break
|
|
3552
|
+
accumulated.extend(chunk)
|
|
3553
|
+
if not early_eof:
|
|
3554
|
+
runtime_boundaries["artifact_content_read_for_whole_file_verification"] = True
|
|
3555
|
+
extra_data = bool(os.read(fd, 1))
|
|
3556
|
+
except OSError:
|
|
3557
|
+
return None, ["artifact_read_failed"]
|
|
3558
|
+
blockers = []
|
|
3559
|
+
if early_eof or extra_data or len(accumulated) != declared_bytes:
|
|
3560
|
+
blockers.extend(["receipt_content_size_mismatch", "artifact_changed_during_read"])
|
|
3561
|
+
blockers = ordered_proof_taxonomy(blockers, PROOF_VERIFICATION_BLOCKER_ORDER)
|
|
3562
|
+
return (None, blockers) if blockers else (bytes(accumulated), [])
|
|
3563
|
+
|
|
3564
|
+
|
|
3565
|
+
def verify_proof_receipt(
|
|
3566
|
+
artifact_fd: int,
|
|
3567
|
+
receipt: str,
|
|
3568
|
+
runtime_boundaries: dict[str, Any],
|
|
3569
|
+
) -> dict[str, Any]:
|
|
3570
|
+
metadata_name = f"{receipt}.json"
|
|
3571
|
+
content_name = f"{receipt}.txt"
|
|
3572
|
+
metadata_precheck, metadata_missing, metadata_blockers = proof_leaf_precheck(
|
|
3573
|
+
artifact_fd, metadata_name, "metadata"
|
|
3574
|
+
)
|
|
3575
|
+
content_precheck, content_missing, content_blockers = proof_leaf_precheck(
|
|
3576
|
+
artifact_fd, content_name, "content"
|
|
3577
|
+
)
|
|
3578
|
+
blockers = [*metadata_blockers, *content_blockers]
|
|
3579
|
+
if metadata_missing or content_missing:
|
|
3580
|
+
blockers.append("receipt_pair_incomplete")
|
|
3581
|
+
blockers = ordered_proof_taxonomy(blockers, PROOF_VERIFICATION_BLOCKER_ORDER)
|
|
3582
|
+
result: dict[str, Any] = {
|
|
3583
|
+
"actual_lines": None,
|
|
3584
|
+
"actual_sha256": None,
|
|
3585
|
+
"blockers": blockers,
|
|
3586
|
+
"content_file_verified": False,
|
|
3587
|
+
"matches_receipt_metadata": False,
|
|
3588
|
+
"metadata_file_verified": False,
|
|
3589
|
+
"metadata_verified": False,
|
|
3590
|
+
"stored_bytes": None,
|
|
3591
|
+
"stored_lines": None,
|
|
3592
|
+
}
|
|
3593
|
+
if blockers or metadata_precheck is None or content_precheck is None:
|
|
3594
|
+
return result
|
|
3595
|
+
|
|
3596
|
+
metadata_fd: int | None = None
|
|
3597
|
+
content_fd: int | None = None
|
|
3598
|
+
metadata_opened: Any | None = None
|
|
3599
|
+
content_opened: Any | None = None
|
|
3600
|
+
try:
|
|
3601
|
+
metadata_fd, metadata_opened, blockers = open_proof_leaf(
|
|
3602
|
+
artifact_fd, metadata_name, metadata_precheck, "metadata"
|
|
3603
|
+
)
|
|
3604
|
+
if blockers or metadata_fd is None or metadata_opened is None:
|
|
3605
|
+
result["blockers"] = blockers
|
|
3606
|
+
return result
|
|
3607
|
+
content_fd, content_opened, blockers = open_proof_leaf(
|
|
3608
|
+
artifact_fd, content_name, content_precheck, "content"
|
|
3609
|
+
)
|
|
3610
|
+
if blockers or content_fd is None or content_opened is None:
|
|
3611
|
+
result["blockers"] = blockers
|
|
3612
|
+
return result
|
|
3613
|
+
|
|
3614
|
+
metadata_raw, blockers = read_proof_metadata_leaf(metadata_fd, metadata_opened)
|
|
3615
|
+
if blockers or metadata_raw is None:
|
|
3616
|
+
result["blockers"] = blockers
|
|
3617
|
+
return result
|
|
3618
|
+
metadata, blockers = decode_proof_receipt_metadata(metadata_raw, receipt)
|
|
3619
|
+
if blockers or metadata is None:
|
|
3620
|
+
result["blockers"] = blockers
|
|
3621
|
+
return result
|
|
3622
|
+
stored = metadata["stored_output"]
|
|
3623
|
+
result.update({
|
|
3624
|
+
"metadata_file_verified": True,
|
|
3625
|
+
"metadata_verified": True,
|
|
3626
|
+
"stored_bytes": stored["bytes"],
|
|
3627
|
+
"stored_lines": stored["lines"],
|
|
3628
|
+
})
|
|
3629
|
+
|
|
3630
|
+
content_raw, blockers = read_proof_content_leaf(
|
|
3631
|
+
content_fd,
|
|
3632
|
+
content_opened,
|
|
3633
|
+
stored["bytes"],
|
|
3634
|
+
runtime_boundaries,
|
|
3635
|
+
)
|
|
3636
|
+
if blockers or content_raw is None:
|
|
3637
|
+
result["blockers"] = blockers
|
|
3638
|
+
return result
|
|
3639
|
+
actual_sha256 = hashlib.sha256(content_raw).hexdigest()
|
|
3640
|
+
actual_lines = content_raw.count(b"\n") + int(
|
|
3641
|
+
bool(content_raw and not content_raw.endswith(b"\n"))
|
|
3642
|
+
)
|
|
3643
|
+
result["actual_sha256"] = actual_sha256
|
|
3644
|
+
result["actual_lines"] = actual_lines
|
|
3645
|
+
result["matches_receipt_metadata"] = actual_sha256 == stored["sha256"]
|
|
3646
|
+
if not result["matches_receipt_metadata"]:
|
|
3647
|
+
blockers.append("receipt_content_hash_mismatch")
|
|
3648
|
+
if actual_lines != stored["lines"]:
|
|
3649
|
+
blockers.append("receipt_line_count_mismatch")
|
|
3650
|
+
result["content_file_verified"] = not blockers
|
|
3651
|
+
result["blockers"] = ordered_proof_taxonomy(
|
|
3652
|
+
blockers, PROOF_VERIFICATION_BLOCKER_ORDER
|
|
3653
|
+
)
|
|
3654
|
+
return result
|
|
3655
|
+
finally:
|
|
3656
|
+
stability_blockers: list[str] = []
|
|
3657
|
+
for fd, opened in (
|
|
3658
|
+
(metadata_fd, metadata_opened),
|
|
3659
|
+
(content_fd, content_opened),
|
|
3660
|
+
):
|
|
3661
|
+
if fd is None:
|
|
3662
|
+
continue
|
|
3663
|
+
try:
|
|
3664
|
+
after = os.fstat(fd)
|
|
3665
|
+
if opened is None or proof_stat_stability(after) != proof_stat_stability(opened):
|
|
3666
|
+
stability_blockers.append("artifact_changed_during_read")
|
|
3667
|
+
except OSError:
|
|
3668
|
+
stability_blockers.append("artifact_read_failed")
|
|
3669
|
+
try:
|
|
3670
|
+
os.close(fd)
|
|
3671
|
+
except OSError:
|
|
3672
|
+
pass
|
|
3673
|
+
if stability_blockers:
|
|
3674
|
+
result["blockers"] = ordered_proof_taxonomy(
|
|
3675
|
+
[*result["blockers"], *stability_blockers],
|
|
3676
|
+
PROOF_VERIFICATION_BLOCKER_ORDER,
|
|
3677
|
+
)
|
|
3678
|
+
result["content_file_verified"] = False
|
|
3679
|
+
result["matches_receipt_metadata"] = False
|
|
3680
|
+
result["metadata_file_verified"] = False
|
|
3681
|
+
result["metadata_verified"] = False
|
|
3682
|
+
|
|
3683
|
+
|
|
3684
|
+
def verify_proof_range_bounds(row: dict[str, Any], stored_bytes: int, stored_lines: int) -> bool:
|
|
3685
|
+
safe_range = row["safe_range"]
|
|
3686
|
+
if safe_range is None:
|
|
3687
|
+
return True
|
|
3688
|
+
if safe_range["kind"] == "lines":
|
|
3689
|
+
passed = 1 <= safe_range["start"] <= safe_range["end"] <= stored_lines
|
|
3690
|
+
else:
|
|
3691
|
+
passed = 0 <= safe_range["start"] < safe_range["end"] <= stored_bytes
|
|
3692
|
+
safe_range["bounds_checked"] = True
|
|
3693
|
+
safe_range["status"] = "verified" if passed else "verification_failed"
|
|
3694
|
+
return passed
|
|
3695
|
+
|
|
3696
|
+
|
|
3697
|
+
def proof_verification_runtime_boundaries() -> dict[str, Any]:
|
|
3698
|
+
return {
|
|
3699
|
+
"artifact_content_read_for_whole_file_verification": False,
|
|
3700
|
+
"command_executed": False,
|
|
3701
|
+
"config_read": False,
|
|
3702
|
+
"content_echoed": False,
|
|
3703
|
+
"current_time_generated": False,
|
|
3704
|
+
"files_written": False,
|
|
3705
|
+
"hosted_savings_claim_allowed": False,
|
|
3706
|
+
"network_or_provider_called": False,
|
|
3707
|
+
"range_content_retrieved": False,
|
|
3708
|
+
"replacement_authorized": False,
|
|
3709
|
+
"source_or_stdin_read": False,
|
|
3710
|
+
"subprocess_started": False,
|
|
3711
|
+
}
|
|
3712
|
+
|
|
3713
|
+
|
|
3714
|
+
def proof_carrying_context_verify_payload(args: argparse.Namespace) -> dict[str, Any]:
|
|
3715
|
+
raw_units = args.proof_unit_json or []
|
|
3716
|
+
supplied_count = len(raw_units)
|
|
3717
|
+
detailed_count = min(supplied_count, PROOF_CARRYING_CONTEXT_DETAILED_UNIT_CAP)
|
|
3718
|
+
overflow_count = max(supplied_count - PROOF_CARRYING_CONTEXT_DETAILED_UNIT_CAP, 0)
|
|
3719
|
+
rows: list[dict[str, Any]] = []
|
|
3720
|
+
plan_rows: list[dict[str, Any] | None] = []
|
|
3721
|
+
conflict_inputs: list[tuple[str | None, str | None]] = []
|
|
3722
|
+
preflight_issues: list[list[str]] = []
|
|
3723
|
+
for unit_index, raw in enumerate(raw_units[:PROOF_CARRYING_CONTEXT_DETAILED_UNIT_CAP]):
|
|
3724
|
+
decoded, terminal_row = decode_proof_unit_json(raw, unit_index)
|
|
3725
|
+
if terminal_row is not None:
|
|
3726
|
+
issues = list(terminal_row["validation_issues"])
|
|
3727
|
+
rows.append(proof_verification_row(unit_index, issues))
|
|
3728
|
+
plan_rows.append(None)
|
|
3729
|
+
conflict_inputs.append((None, None))
|
|
3730
|
+
preflight_issues.append(issues)
|
|
3731
|
+
continue
|
|
3732
|
+
assert decoded is not None
|
|
3733
|
+
plan_row, receipt, declared_hash = normalize_proof_unit(decoded, unit_index)
|
|
3734
|
+
issues = list(plan_row["validation_issues"])
|
|
3735
|
+
rows.append(
|
|
3736
|
+
proof_verification_row(unit_index, issues)
|
|
3737
|
+
if issues
|
|
3738
|
+
else normalized_proof_verification_row(plan_row)
|
|
3739
|
+
)
|
|
3740
|
+
plan_rows.append(plan_row)
|
|
3741
|
+
conflict_inputs.append((receipt, declared_hash))
|
|
3742
|
+
preflight_issues.append(issues)
|
|
3743
|
+
|
|
3744
|
+
hashes_by_receipt: dict[str, set[str]] = {}
|
|
3745
|
+
for receipt, declared_hash in conflict_inputs:
|
|
3746
|
+
if receipt is not None and declared_hash is not None:
|
|
3747
|
+
hashes_by_receipt.setdefault(receipt, set()).add(declared_hash)
|
|
3748
|
+
conflicted_receipts = {
|
|
3749
|
+
receipt for receipt, hashes in hashes_by_receipt.items() if len(hashes) > 1
|
|
3750
|
+
}
|
|
3751
|
+
for index, (receipt, _declared_hash) in enumerate(conflict_inputs):
|
|
3752
|
+
if receipt in conflicted_receipts:
|
|
3753
|
+
issues = ordered_proof_taxonomy(
|
|
3754
|
+
[*preflight_issues[index], "receipt_hash_conflict"],
|
|
3755
|
+
PROOF_VERIFICATION_BLOCKER_ORDER,
|
|
3756
|
+
)
|
|
3757
|
+
preflight_issues[index] = issues
|
|
3758
|
+
rows[index] = proof_verification_row(index, issues)
|
|
3759
|
+
|
|
3760
|
+
duplicate_groups: dict[tuple[Any, ...], list[int]] = {}
|
|
3761
|
+
for index, plan_row in enumerate(plan_rows):
|
|
3762
|
+
if plan_row is not None and not preflight_issues[index]:
|
|
3763
|
+
duplicate_groups.setdefault(proof_duplicate_key(plan_row), []).append(index)
|
|
3764
|
+
duplicate_indexes = {
|
|
3765
|
+
index
|
|
3766
|
+
for indexes in duplicate_groups.values()
|
|
3767
|
+
if len(indexes) > 1
|
|
3768
|
+
for index in indexes
|
|
3769
|
+
}
|
|
3770
|
+
for index, row in enumerate(rows):
|
|
3771
|
+
if preflight_issues[index]:
|
|
3772
|
+
continue
|
|
3773
|
+
warnings = list(PROOF_VERIFICATION_WARNING_ORDER[:3])
|
|
3774
|
+
if row["safe_range"] is None:
|
|
3775
|
+
warnings.append("safe_range_not_supplied")
|
|
3776
|
+
if index in duplicate_indexes:
|
|
3777
|
+
warnings.append("duplicate_proof_unit")
|
|
3778
|
+
row["warnings"] = ordered_proof_taxonomy(
|
|
3779
|
+
warnings, PROOF_VERIFICATION_WARNING_ORDER
|
|
3780
|
+
)
|
|
3781
|
+
|
|
3782
|
+
request_blockers: list[str] = []
|
|
3783
|
+
if supplied_count == 0:
|
|
3784
|
+
request_blockers.append("missing_proof_unit")
|
|
3785
|
+
if overflow_count:
|
|
3786
|
+
request_blockers.append("too_many_proof_units")
|
|
3787
|
+
for issues in preflight_issues:
|
|
3788
|
+
request_blockers.extend(issues)
|
|
3789
|
+
|
|
3790
|
+
normalized_dir, directory_issue = validate_proof_artifact_dir_arg(
|
|
3791
|
+
getattr(args, "artifact_dir", None)
|
|
3792
|
+
)
|
|
3793
|
+
if directory_issue:
|
|
3794
|
+
request_blockers.append(directory_issue)
|
|
3795
|
+
if not proof_artifact_io_capabilities_available():
|
|
3796
|
+
request_blockers.append("artifact_io_capability_unavailable")
|
|
3797
|
+
preflight_aborted = bool(request_blockers)
|
|
3798
|
+
if preflight_aborted:
|
|
3799
|
+
for index, row in enumerate(rows):
|
|
3800
|
+
if not preflight_issues[index]:
|
|
3801
|
+
row["blockers"] = ["request_preflight_aborted"]
|
|
3802
|
+
if rows and any(not issues for issues in preflight_issues):
|
|
3803
|
+
request_blockers.append("request_preflight_aborted")
|
|
3804
|
+
|
|
3805
|
+
runtime_boundaries = proof_verification_runtime_boundaries()
|
|
3806
|
+
artifact_fd: int | None = None
|
|
3807
|
+
directory_stage_issue: str | None = None
|
|
3808
|
+
if not preflight_aborted:
|
|
3809
|
+
assert normalized_dir is not None
|
|
3810
|
+
artifact_fd, directory_stage_issue = open_proof_artifact_directory(normalized_dir)
|
|
3811
|
+
if directory_stage_issue:
|
|
3812
|
+
for row in rows:
|
|
3813
|
+
row["blockers"] = [directory_stage_issue]
|
|
3814
|
+
request_blockers.append(directory_stage_issue)
|
|
3815
|
+
try:
|
|
3816
|
+
if artifact_fd is not None:
|
|
3817
|
+
receipt_cache: dict[str, dict[str, Any]] = {}
|
|
3818
|
+
for row in rows:
|
|
3819
|
+
receipt = row["receipt"]["id"]
|
|
3820
|
+
assert isinstance(receipt, str)
|
|
3821
|
+
if receipt not in receipt_cache:
|
|
3822
|
+
receipt_cache[receipt] = verify_proof_receipt(
|
|
3823
|
+
artifact_fd, receipt, runtime_boundaries
|
|
3824
|
+
)
|
|
3825
|
+
result = receipt_cache[receipt]
|
|
3826
|
+
row["blockers"] = list(result["blockers"])
|
|
3827
|
+
row["receipt"].update({
|
|
3828
|
+
"content_file_verified": result["content_file_verified"],
|
|
3829
|
+
"metadata_file_verified": result["metadata_file_verified"],
|
|
3830
|
+
"metadata_verified": result["metadata_verified"],
|
|
3831
|
+
"stored_bytes": result["stored_bytes"],
|
|
3832
|
+
"stored_lines": result["stored_lines"],
|
|
3833
|
+
"verified": bool(
|
|
3834
|
+
result["metadata_verified"] and result["content_file_verified"]
|
|
3835
|
+
),
|
|
3836
|
+
})
|
|
3837
|
+
if result["actual_sha256"] is not None:
|
|
3838
|
+
matches_proof = (
|
|
3839
|
+
result["actual_sha256"] == row["content_hash"]["declared_value"]
|
|
3840
|
+
)
|
|
3841
|
+
row["content_hash"].update({
|
|
3842
|
+
"matches_proof_unit": matches_proof,
|
|
3843
|
+
"matches_receipt_metadata": result["matches_receipt_metadata"],
|
|
3844
|
+
"verified": bool(matches_proof and result["matches_receipt_metadata"]),
|
|
3845
|
+
})
|
|
3846
|
+
if not matches_proof:
|
|
3847
|
+
row["blockers"].append("proof_content_hash_mismatch")
|
|
3848
|
+
if (
|
|
3849
|
+
row["receipt"]["verified"]
|
|
3850
|
+
and row["content_hash"]["verified"]
|
|
3851
|
+
and not verify_proof_range_bounds(
|
|
3852
|
+
row, result["stored_bytes"], result["stored_lines"]
|
|
3853
|
+
)
|
|
3854
|
+
):
|
|
3855
|
+
row["blockers"].append("safe_range_out_of_bounds")
|
|
3856
|
+
row["blockers"] = ordered_proof_taxonomy(
|
|
3857
|
+
row["blockers"], PROOF_VERIFICATION_BLOCKER_ORDER
|
|
3858
|
+
)
|
|
3859
|
+
if not row["blockers"]:
|
|
3860
|
+
row["status"] = "verified"
|
|
3861
|
+
finally:
|
|
3862
|
+
if artifact_fd is not None:
|
|
3863
|
+
try:
|
|
3864
|
+
os.close(artifact_fd)
|
|
3865
|
+
except OSError:
|
|
3866
|
+
pass
|
|
3867
|
+
|
|
3868
|
+
top_blockers = list(request_blockers)
|
|
3869
|
+
for row in rows:
|
|
3870
|
+
top_blockers.extend(row["blockers"])
|
|
3871
|
+
top_blockers = ordered_proof_taxonomy(
|
|
3872
|
+
top_blockers, PROOF_VERIFICATION_BLOCKER_ORDER
|
|
3873
|
+
)
|
|
3874
|
+
verified_count = sum(row["status"] == "verified" for row in rows)
|
|
3875
|
+
failed_count = supplied_count - verified_count
|
|
3876
|
+
unique_receipt_count = len({receipt for receipt, _hash in conflict_inputs if receipt})
|
|
3877
|
+
status = (
|
|
3878
|
+
"verified"
|
|
3879
|
+
if supplied_count > 0
|
|
3880
|
+
and overflow_count == 0
|
|
3881
|
+
and verified_count == detailed_count
|
|
3882
|
+
and not top_blockers
|
|
3883
|
+
else "verification_failed"
|
|
3884
|
+
)
|
|
3885
|
+
return {
|
|
3886
|
+
"artifact_scope": {
|
|
3887
|
+
"content_byte_cap": PROOF_RECEIPT_CONTENT_BYTE_CAP,
|
|
3888
|
+
"directory_echoed": False,
|
|
3889
|
+
"exact_receipt_files_only": True,
|
|
3890
|
+
"explicit_directory": True,
|
|
3891
|
+
"fallback_directories_searched": False,
|
|
3892
|
+
"metadata_byte_cap": PROOF_RECEIPT_METADATA_BYTE_CAP,
|
|
3893
|
+
"posix_private_mode_required": True,
|
|
3894
|
+
"same_effective_owner_required": True,
|
|
3895
|
+
"symlinks_followed": False,
|
|
3896
|
+
},
|
|
3897
|
+
"blocker_order": list(PROOF_VERIFICATION_BLOCKER_ORDER),
|
|
3898
|
+
"blockers": top_blockers,
|
|
3899
|
+
"candidate_replacement": None,
|
|
3900
|
+
"claim_boundary": PROOF_VERIFICATION_CLAIM_BOUNDARY,
|
|
3901
|
+
"experiment_id": "proof-carrying-context",
|
|
3902
|
+
"mode": "verify",
|
|
3903
|
+
"process_exit_contract": PROOF_VERIFICATION_PROCESS_EXIT_CONTRACT,
|
|
3904
|
+
"proof_unit_schema_version": PROOF_CARRYING_CONTEXT_UNIT_SCHEMA_VERSION,
|
|
3905
|
+
"proof_units": rows,
|
|
3906
|
+
"runtime_boundaries": runtime_boundaries,
|
|
3907
|
+
"schema": PROOF_CARRYING_CONTEXT_VERIFY_SCHEMA_VERSION,
|
|
3908
|
+
"status": status,
|
|
3909
|
+
"summary": {
|
|
3910
|
+
"detailed_unit_count": detailed_count,
|
|
3911
|
+
"failed_unit_count": failed_count,
|
|
3912
|
+
"overflow_unit_count": overflow_count,
|
|
3913
|
+
"supplied_unit_count": supplied_count,
|
|
3914
|
+
"unique_receipt_count": unique_receipt_count,
|
|
3915
|
+
"verified_unit_count": verified_count,
|
|
3916
|
+
},
|
|
3917
|
+
"warning_order": list(PROOF_VERIFICATION_WARNING_ORDER),
|
|
3918
|
+
}
|
|
3919
|
+
|
|
3920
|
+
|
|
3921
|
+
def command_verify_proof_carrying_context(args: argparse.Namespace) -> int:
|
|
3922
|
+
payload = proof_carrying_context_verify_payload(args)
|
|
3923
|
+
if args.json:
|
|
3924
|
+
sys.stdout.write(json.dumps(payload, ensure_ascii=False, indent=2, sort_keys=True) + "\n")
|
|
3925
|
+
else:
|
|
3926
|
+
print("ContextGuard proof-carrying-context verification")
|
|
3927
|
+
print(f"Status: {payload['status']}")
|
|
3928
|
+
for key in (
|
|
3929
|
+
"supplied_unit_count",
|
|
3930
|
+
"detailed_unit_count",
|
|
3931
|
+
"overflow_unit_count",
|
|
3932
|
+
"unique_receipt_count",
|
|
3933
|
+
"verified_unit_count",
|
|
3934
|
+
"failed_unit_count",
|
|
3935
|
+
):
|
|
3936
|
+
print(f"{key}: {payload['summary'][key]}")
|
|
3937
|
+
print(f"Blockers: {', '.join(payload['blockers'])}")
|
|
3938
|
+
print(payload["claim_boundary"])
|
|
3939
|
+
return 0 if payload["status"] == "verified" else 2
|
|
3940
|
+
|
|
3941
|
+
|
|
3942
|
+
_SEMANTIC_GC_NONFINITE_SENTINEL = object()
|
|
3943
|
+
|
|
3944
|
+
|
|
3945
|
+
def ordered_semantic_gc_taxonomy(values: list[str] | set[str], order: tuple[str, ...]) -> list[str]:
|
|
3946
|
+
selected = set(values)
|
|
3947
|
+
return [value for value in order if value in selected]
|
|
3948
|
+
|
|
3949
|
+
|
|
3950
|
+
def semantic_gc_validation_row(index: int, issue: str | None = None) -> dict[str, Any]:
|
|
3951
|
+
return {
|
|
3952
|
+
"candidate_safety_applicable": None,
|
|
3953
|
+
"candidate_safety_issues": [],
|
|
3954
|
+
"input_index": index,
|
|
3955
|
+
"structural_issues": [issue] if issue else [],
|
|
3956
|
+
"unit_id": None,
|
|
3957
|
+
}
|
|
3958
|
+
|
|
3959
|
+
|
|
3960
|
+
def decode_semantic_gc_unit(raw: Any, index: int) -> tuple[dict[str, Any] | None, dict[str, Any]]:
|
|
3961
|
+
row = semantic_gc_validation_row(index)
|
|
3962
|
+
if not isinstance(raw, str):
|
|
3963
|
+
row["structural_issues"] = ["invalid_context_unit_json"]
|
|
3964
|
+
return None, row
|
|
3965
|
+
try:
|
|
3966
|
+
encoded = raw.encode("utf-8", errors="strict")
|
|
3967
|
+
except UnicodeEncodeError:
|
|
3968
|
+
row["structural_issues"] = ["invalid_unicode_scalar"]
|
|
3969
|
+
return None, row
|
|
3970
|
+
if len(encoded) > SEMANTIC_GC_UNIT_JSON_BYTE_CAP:
|
|
3971
|
+
row["structural_issues"] = ["invalid_context_unit_json"]
|
|
3972
|
+
return None, row
|
|
3973
|
+
|
|
3974
|
+
duplicate_key = False
|
|
3975
|
+
|
|
3976
|
+
def pairs_hook(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
|
|
3977
|
+
nonlocal duplicate_key
|
|
3978
|
+
result: dict[str, Any] = {}
|
|
3979
|
+
for key, value in pairs:
|
|
3980
|
+
if key in result:
|
|
3981
|
+
duplicate_key = True
|
|
3982
|
+
result[key] = value
|
|
3983
|
+
return result
|
|
3984
|
+
|
|
3985
|
+
try:
|
|
3986
|
+
decoded = json.loads(
|
|
3987
|
+
raw,
|
|
3988
|
+
object_pairs_hook=pairs_hook,
|
|
3989
|
+
parse_constant=lambda _value: _SEMANTIC_GC_NONFINITE_SENTINEL,
|
|
3990
|
+
)
|
|
3991
|
+
except RecursionError:
|
|
3992
|
+
row["structural_issues"] = ["decoder_recursion_limit"]
|
|
3993
|
+
return None, row
|
|
3994
|
+
except (json.JSONDecodeError, UnicodeDecodeError, ValueError, TypeError):
|
|
3995
|
+
row["structural_issues"] = ["invalid_context_unit_json"]
|
|
3996
|
+
return None, row
|
|
3997
|
+
|
|
3998
|
+
depth_exceeded = False
|
|
3999
|
+
nonfinite = False
|
|
4000
|
+
invalid_unicode = False
|
|
4001
|
+
stack: list[tuple[Any, int]] = [(decoded, 0)]
|
|
4002
|
+
while stack:
|
|
4003
|
+
value, depth = stack.pop()
|
|
4004
|
+
if depth > SEMANTIC_GC_JSON_MAX_DEPTH:
|
|
4005
|
+
depth_exceeded = True
|
|
4006
|
+
continue
|
|
4007
|
+
if isinstance(value, str):
|
|
4008
|
+
try:
|
|
4009
|
+
value.encode("utf-8", errors="strict")
|
|
4010
|
+
except UnicodeEncodeError:
|
|
4011
|
+
invalid_unicode = True
|
|
4012
|
+
elif value is _SEMANTIC_GC_NONFINITE_SENTINEL or (type(value) is float and not math.isfinite(value)):
|
|
4013
|
+
nonfinite = True
|
|
4014
|
+
if isinstance(value, dict):
|
|
4015
|
+
for key, child in value.items():
|
|
4016
|
+
stack.append((key, depth + 1))
|
|
4017
|
+
stack.append((child, depth + 1))
|
|
4018
|
+
elif isinstance(value, list):
|
|
4019
|
+
for child in value:
|
|
4020
|
+
stack.append((child, depth + 1))
|
|
4021
|
+
|
|
4022
|
+
issue = None
|
|
4023
|
+
if duplicate_key:
|
|
4024
|
+
issue = "duplicate_json_key"
|
|
4025
|
+
elif depth_exceeded:
|
|
4026
|
+
issue = "context_unit_depth_exceeded"
|
|
4027
|
+
elif nonfinite:
|
|
4028
|
+
issue = "nonfinite_json_number"
|
|
4029
|
+
elif invalid_unicode:
|
|
4030
|
+
issue = "invalid_unicode_scalar"
|
|
4031
|
+
elif not isinstance(decoded, dict):
|
|
4032
|
+
issue = "invalid_context_unit_json"
|
|
4033
|
+
if issue:
|
|
4034
|
+
row["structural_issues"] = [issue]
|
|
4035
|
+
return None, row
|
|
4036
|
+
return decoded, row
|
|
4037
|
+
|
|
4038
|
+
|
|
4039
|
+
def normalize_semantic_gc_structure(obj: dict[str, Any], row: dict[str, Any]) -> dict[str, Any]:
|
|
4040
|
+
issues: list[str] = []
|
|
4041
|
+
if obj.get("schema") != SEMANTIC_GC_UNIT_SCHEMA_VERSION:
|
|
4042
|
+
issues.append("invalid_context_unit_schema")
|
|
4043
|
+
if set(obj) - SEMANTIC_GC_ALLOWED_FIELDS:
|
|
4044
|
+
issues.append("unknown_context_unit_field")
|
|
4045
|
+
|
|
4046
|
+
raw_id = obj.get("unit_id")
|
|
4047
|
+
unit_id = raw_id if isinstance(raw_id, str) and SEMANTIC_GC_UNIT_ID_RE.fullmatch(raw_id) else None
|
|
4048
|
+
if raw_id is None or raw_id == "":
|
|
4049
|
+
issues.append("missing_unit_id")
|
|
4050
|
+
elif unit_id is None:
|
|
4051
|
+
issues.append("invalid_unit_id")
|
|
4052
|
+
row["unit_id"] = unit_id
|
|
4053
|
+
|
|
4054
|
+
raw_references = obj.get("references")
|
|
4055
|
+
references: list[str] = []
|
|
4056
|
+
reference_targets: list[str] = []
|
|
4057
|
+
if not isinstance(raw_references, list) or len(raw_references) > 64:
|
|
4058
|
+
issues.append("invalid_references")
|
|
4059
|
+
else:
|
|
4060
|
+
seen: set[str] = set()
|
|
4061
|
+
reference_issue = False
|
|
4062
|
+
for reference in raw_references:
|
|
4063
|
+
if not isinstance(reference, str) or SEMANTIC_GC_UNIT_ID_RE.fullmatch(reference) is None:
|
|
4064
|
+
issues.append("invalid_references")
|
|
4065
|
+
reference_issue = True
|
|
4066
|
+
continue
|
|
4067
|
+
if reference in seen:
|
|
4068
|
+
issues.append("duplicate_reference")
|
|
4069
|
+
reference_issue = True
|
|
4070
|
+
continue
|
|
4071
|
+
references.append(reference)
|
|
4072
|
+
reference_targets.append(reference)
|
|
4073
|
+
seen.add(reference)
|
|
4074
|
+
if reference_issue:
|
|
4075
|
+
references = []
|
|
4076
|
+
|
|
4077
|
+
is_root = obj.get("is_root")
|
|
4078
|
+
if type(is_root) is not bool:
|
|
4079
|
+
issues.append("invalid_root_flag")
|
|
4080
|
+
is_root = None
|
|
4081
|
+
protected = obj.get("protected_zone")
|
|
4082
|
+
if type(protected) is not bool:
|
|
4083
|
+
issues.append("invalid_protected_zone_flag")
|
|
4084
|
+
protected = None
|
|
4085
|
+
row["structural_issues"] = ordered_semantic_gc_taxonomy(issues, SEMANTIC_GC_BLOCKER_ORDER)
|
|
4086
|
+
return {
|
|
4087
|
+
"object": obj,
|
|
4088
|
+
"row": row,
|
|
4089
|
+
"unit_id": unit_id,
|
|
4090
|
+
"references": references,
|
|
4091
|
+
"reference_targets": reference_targets,
|
|
4092
|
+
"is_root": is_root,
|
|
4093
|
+
"protected_zone": protected,
|
|
4094
|
+
}
|
|
4095
|
+
|
|
4096
|
+
|
|
4097
|
+
def valid_semantic_gc_note(value: Any) -> bool:
|
|
4098
|
+
if not isinstance(value, str) or not (1 <= len(value) <= 512) or value != value.strip():
|
|
4099
|
+
return False
|
|
4100
|
+
return not any(unicodedata.category(char) in {"Cc", "Cf", "Zl", "Zp"} for char in value)
|
|
4101
|
+
|
|
4102
|
+
|
|
4103
|
+
def normalize_semantic_gc_candidate(unit: dict[str, Any]) -> tuple[dict[str, Any], list[str], str | None, str | None]:
|
|
4104
|
+
obj = unit["object"]
|
|
4105
|
+
issues: list[str] = []
|
|
4106
|
+
raw_hash = obj.get("content_sha256")
|
|
4107
|
+
content_hash = raw_hash if isinstance(raw_hash, str) and SEMANTIC_GC_CONTENT_SHA256_RE.fullmatch(raw_hash) else None
|
|
4108
|
+
if content_hash is None:
|
|
4109
|
+
issues.append("invalid_content_sha256")
|
|
4110
|
+
|
|
4111
|
+
raw_provenance = obj.get("provenance")
|
|
4112
|
+
source_label = None
|
|
4113
|
+
receipt_id = None
|
|
4114
|
+
if raw_provenance is None:
|
|
4115
|
+
issues.append("missing_provenance")
|
|
4116
|
+
elif not isinstance(raw_provenance, dict) or set(raw_provenance) != {"source_label", "receipt_id"}:
|
|
4117
|
+
issues.append("invalid_provenance")
|
|
4118
|
+
else:
|
|
4119
|
+
raw_label = raw_provenance.get("source_label")
|
|
4120
|
+
if (
|
|
4121
|
+
isinstance(raw_label, str)
|
|
4122
|
+
and raw_label == raw_label.strip()
|
|
4123
|
+
and SEMANTIC_GC_SOURCE_LABEL_RE.fullmatch(raw_label) is not None
|
|
4124
|
+
):
|
|
4125
|
+
source_label = raw_label
|
|
4126
|
+
else:
|
|
4127
|
+
issues.append("invalid_source_label")
|
|
4128
|
+
raw_receipt = raw_provenance.get("receipt_id")
|
|
4129
|
+
if isinstance(raw_receipt, str) and SEMANTIC_GC_RECEIPT_ID_RE.fullmatch(raw_receipt):
|
|
4130
|
+
receipt_id = raw_receipt
|
|
4131
|
+
else:
|
|
4132
|
+
issues.append("invalid_receipt_id")
|
|
4133
|
+
|
|
4134
|
+
raw_note = obj.get("missed_context_note")
|
|
4135
|
+
note = raw_note if valid_semantic_gc_note(raw_note) else None
|
|
4136
|
+
if raw_note is None or raw_note == "":
|
|
4137
|
+
issues.append("missing_missed_context_note")
|
|
4138
|
+
elif note is None:
|
|
4139
|
+
issues.append("invalid_missed_context_note")
|
|
4140
|
+
|
|
4141
|
+
raw_fallback = obj.get("exact_fallback_command")
|
|
4142
|
+
fallback = None
|
|
4143
|
+
if raw_fallback is None or raw_fallback == "":
|
|
4144
|
+
issues.append("missing_exact_fallback")
|
|
4145
|
+
elif not isinstance(raw_fallback, str):
|
|
4146
|
+
issues.append("invalid_exact_fallback")
|
|
4147
|
+
else:
|
|
4148
|
+
forbidden = (';', '|', '&', '>', '<', '`', '$', '\\', '\n', '\r', '"', "'")
|
|
4149
|
+
if any(token in raw_fallback for token in forbidden) or raw_fallback != raw_fallback.strip():
|
|
4150
|
+
issues.append("invalid_exact_fallback")
|
|
4151
|
+
else:
|
|
4152
|
+
parts = raw_fallback.split(" ")
|
|
4153
|
+
if len(parts) != 4 or parts[0] != "context-guard-artifact" or parts[1] != "get" or parts[3] != "--full":
|
|
4154
|
+
issues.append("invalid_exact_fallback")
|
|
4155
|
+
elif SEMANTIC_GC_RECEIPT_ID_RE.fullmatch(parts[2]) is None:
|
|
4156
|
+
issues.append("invalid_exact_fallback")
|
|
4157
|
+
elif receipt_id is not None and parts[2] != receipt_id:
|
|
4158
|
+
issues.append("fallback_receipt_mismatch")
|
|
4159
|
+
elif receipt_id is None:
|
|
4160
|
+
issues.append("invalid_exact_fallback")
|
|
4161
|
+
else:
|
|
4162
|
+
fallback = raw_fallback
|
|
4163
|
+
|
|
4164
|
+
issues = ordered_semantic_gc_taxonomy(issues, SEMANTIC_GC_BLOCKER_ORDER)
|
|
4165
|
+
return ({
|
|
4166
|
+
"candidate_replacement": None,
|
|
4167
|
+
"candidate_safety_issues": issues,
|
|
4168
|
+
"content_sha256": content_hash,
|
|
4169
|
+
"exact_fallback_command": fallback,
|
|
4170
|
+
"human_review_required": True,
|
|
4171
|
+
"missed_context_note": note,
|
|
4172
|
+
"protected_zone": False,
|
|
4173
|
+
"provenance": {"receipt_id": receipt_id, "source_label": source_label},
|
|
4174
|
+
"reason": "unreachable_from_declared_roots",
|
|
4175
|
+
"unit_id": unit["unit_id"],
|
|
4176
|
+
}, issues, content_hash, receipt_id)
|
|
4177
|
+
|
|
4178
|
+
|
|
4179
|
+
def semantic_gc_plan_payload(args: argparse.Namespace) -> dict[str, Any]:
|
|
4180
|
+
raw_units = args.context_unit_json or []
|
|
4181
|
+
total_count = len(raw_units)
|
|
4182
|
+
detailed_count = min(total_count, SEMANTIC_GC_DETAILED_UNIT_CAP)
|
|
4183
|
+
overflow_count = max(total_count - SEMANTIC_GC_DETAILED_UNIT_CAP, 0)
|
|
4184
|
+
units: list[dict[str, Any]] = []
|
|
4185
|
+
rows: list[dict[str, Any]] = []
|
|
4186
|
+
decoded_count = 0
|
|
4187
|
+
for index, raw in enumerate(raw_units[:SEMANTIC_GC_DETAILED_UNIT_CAP]):
|
|
4188
|
+
decoded, row = decode_semantic_gc_unit(raw, index)
|
|
4189
|
+
rows.append(row)
|
|
4190
|
+
if decoded is None:
|
|
4191
|
+
continue
|
|
4192
|
+
decoded_count += 1
|
|
4193
|
+
units.append(normalize_semantic_gc_structure(decoded, row))
|
|
4194
|
+
|
|
4195
|
+
by_id: dict[str, list[dict[str, Any]]] = {}
|
|
4196
|
+
for unit in units:
|
|
4197
|
+
if unit["unit_id"] is not None:
|
|
4198
|
+
by_id.setdefault(unit["unit_id"], []).append(unit)
|
|
4199
|
+
duplicate_ids = {unit_id for unit_id, matches in by_id.items() if len(matches) > 1}
|
|
4200
|
+
for unit in units:
|
|
4201
|
+
structural = list(unit["row"]["structural_issues"])
|
|
4202
|
+
if unit["unit_id"] in duplicate_ids:
|
|
4203
|
+
structural.append("duplicate_unit_id")
|
|
4204
|
+
if any(reference not in by_id for reference in unit["reference_targets"]):
|
|
4205
|
+
structural.append("unknown_reference")
|
|
4206
|
+
if any(reference in duplicate_ids for reference in unit["reference_targets"]):
|
|
4207
|
+
structural.append("ambiguous_reference")
|
|
4208
|
+
unit["row"]["structural_issues"] = ordered_semantic_gc_taxonomy(structural, SEMANTIC_GC_BLOCKER_ORDER)
|
|
4209
|
+
|
|
4210
|
+
declared_roots = sorted(
|
|
4211
|
+
unit_id for unit_id, matches in by_id.items()
|
|
4212
|
+
if unit_id not in duplicate_ids and len(matches) == 1 and matches[0]["is_root"] is True
|
|
4213
|
+
)
|
|
4214
|
+
structural_blockers: list[str] = []
|
|
4215
|
+
if total_count == 0:
|
|
4216
|
+
structural_blockers.append("no_context_units")
|
|
4217
|
+
if overflow_count:
|
|
4218
|
+
structural_blockers.append("unit_limit_exceeded")
|
|
4219
|
+
for row in rows:
|
|
4220
|
+
structural_blockers.extend(row["structural_issues"])
|
|
4221
|
+
if not declared_roots:
|
|
4222
|
+
structural_blockers.append("no_declared_root")
|
|
4223
|
+
graph_complete = not structural_blockers
|
|
4224
|
+
|
|
4225
|
+
marked_ids: list[str] = []
|
|
4226
|
+
candidates: list[dict[str, Any]] = []
|
|
4227
|
+
unreachable_count = 0
|
|
4228
|
+
protected_unreachable_count = 0
|
|
4229
|
+
safety_valid_count = 0
|
|
4230
|
+
safety_invalid_count = 0
|
|
4231
|
+
safety_blockers: list[str] = []
|
|
4232
|
+
candidate_hashes: list[str] = []
|
|
4233
|
+
candidate_receipts: list[str] = []
|
|
4234
|
+
if graph_complete:
|
|
4235
|
+
marked: set[str] = set()
|
|
4236
|
+
pending = list(declared_roots)
|
|
4237
|
+
while pending:
|
|
4238
|
+
unit_id = pending.pop()
|
|
4239
|
+
if unit_id in marked:
|
|
4240
|
+
continue
|
|
4241
|
+
marked.add(unit_id)
|
|
4242
|
+
pending.extend(by_id[unit_id][0]["references"])
|
|
4243
|
+
marked_ids = sorted(marked)
|
|
4244
|
+
for unit in units:
|
|
4245
|
+
unit_id = unit["unit_id"]
|
|
4246
|
+
assert unit_id is not None
|
|
4247
|
+
row = unit["row"]
|
|
4248
|
+
if unit_id in marked:
|
|
4249
|
+
row["candidate_safety_applicable"] = False
|
|
4250
|
+
continue
|
|
4251
|
+
unreachable_count += 1
|
|
4252
|
+
if unit["protected_zone"] is True:
|
|
4253
|
+
protected_unreachable_count += 1
|
|
4254
|
+
row["candidate_safety_applicable"] = False
|
|
4255
|
+
continue
|
|
4256
|
+
row["candidate_safety_applicable"] = True
|
|
4257
|
+
candidate, issues, content_hash, receipt_id = normalize_semantic_gc_candidate(unit)
|
|
4258
|
+
row["candidate_safety_issues"] = issues
|
|
4259
|
+
candidates.append(candidate)
|
|
4260
|
+
safety_blockers.extend(issues)
|
|
4261
|
+
if issues:
|
|
4262
|
+
safety_invalid_count += 1
|
|
4263
|
+
else:
|
|
4264
|
+
safety_valid_count += 1
|
|
4265
|
+
if content_hash is not None:
|
|
4266
|
+
candidate_hashes.append(content_hash)
|
|
4267
|
+
if receipt_id is not None:
|
|
4268
|
+
candidate_receipts.append(receipt_id)
|
|
4269
|
+
candidates.sort(key=lambda item: item["unit_id"])
|
|
4270
|
+
|
|
4271
|
+
blockers = list(structural_blockers)
|
|
4272
|
+
if not graph_complete:
|
|
4273
|
+
blockers.append("graph_evaluation_suppressed")
|
|
4274
|
+
protected_policy = "deny" if getattr(args, "protected_zone_policy", None) == "deny" else None
|
|
4275
|
+
if protected_policy is None:
|
|
4276
|
+
blockers.append("protected_zone_policy_required")
|
|
4277
|
+
blockers.extend(safety_blockers)
|
|
4278
|
+
if graph_complete and not args.provider_boundary_ack:
|
|
4279
|
+
blockers.append("provider_boundary_ack_required")
|
|
4280
|
+
if graph_complete and candidates and not args.human_review_ack:
|
|
4281
|
+
blockers.append("human_review_ack_required")
|
|
4282
|
+
blockers = ordered_semantic_gc_taxonomy(blockers, SEMANTIC_GC_BLOCKER_ORDER)
|
|
4283
|
+
|
|
4284
|
+
warnings = list(SEMANTIC_GC_WARNING_ORDER[:6])
|
|
4285
|
+
if candidates:
|
|
4286
|
+
warnings.extend(("human_review_still_required", "accepted_notes_are_untrusted"))
|
|
4287
|
+
if len(candidate_hashes) != len(set(candidate_hashes)):
|
|
4288
|
+
warnings.append("duplicate_content_sha256")
|
|
4289
|
+
if len(candidate_receipts) != len(set(candidate_receipts)):
|
|
4290
|
+
warnings.append("duplicate_receipt_id")
|
|
4291
|
+
if protected_unreachable_count:
|
|
4292
|
+
warnings.append("protected_unreachable_excluded")
|
|
4293
|
+
if graph_complete and not candidates:
|
|
4294
|
+
warnings.append("no_sweep_candidates")
|
|
4295
|
+
warnings = ordered_semantic_gc_taxonomy(warnings, SEMANTIC_GC_WARNING_ORDER)
|
|
4296
|
+
|
|
4297
|
+
verification_scope = {
|
|
4298
|
+
"artifact_content_read": False,
|
|
4299
|
+
"context_content_read": False,
|
|
4300
|
+
"deletion_or_omission_performed": False,
|
|
4301
|
+
"exact_fallback_executed": False,
|
|
4302
|
+
"files_written": False,
|
|
4303
|
+
"model_or_provider_called": False,
|
|
4304
|
+
"network_called": False,
|
|
4305
|
+
"provenance_verified_externally": False,
|
|
4306
|
+
"subprocess_started": False,
|
|
4307
|
+
}
|
|
4308
|
+
return {
|
|
4309
|
+
"blockers": blockers,
|
|
4310
|
+
"candidate_count": len(candidates),
|
|
4311
|
+
"candidate_replacement": None,
|
|
4312
|
+
"candidate_safety_invalid_count": safety_invalid_count,
|
|
4313
|
+
"candidate_safety_valid_count": safety_valid_count,
|
|
4314
|
+
"candidates": candidates,
|
|
4315
|
+
"declared_root_count": len(declared_roots),
|
|
4316
|
+
"declared_root_ids": declared_roots,
|
|
4317
|
+
"decoded_unit_count": decoded_count,
|
|
4318
|
+
"detailed_unit_count": detailed_count,
|
|
4319
|
+
"effective_protected_zone_policy": "deny",
|
|
4320
|
+
"experiment": "semantic-gc",
|
|
4321
|
+
"graph_evaluation_performed": graph_complete,
|
|
4322
|
+
"graph_integrity_complete": graph_complete,
|
|
4323
|
+
"human_review_acknowledged": bool(args.human_review_ack),
|
|
4324
|
+
"human_review_performed": False,
|
|
4325
|
+
"marked_unit_count": len(marked_ids),
|
|
4326
|
+
"marked_unit_ids": marked_ids,
|
|
4327
|
+
"omission_authorized": False,
|
|
4328
|
+
"overflow_unit_count": overflow_count,
|
|
4329
|
+
"plan_only": True,
|
|
4330
|
+
"process_exit_contract": SEMANTIC_GC_PROCESS_EXIT_CONTRACT,
|
|
4331
|
+
"protected_unreachable_count": protected_unreachable_count,
|
|
4332
|
+
"protected_zone_policy": protected_policy,
|
|
4333
|
+
"provider_boundary_acknowledged": bool(args.provider_boundary_ack),
|
|
4334
|
+
"runtime_action_allowed": False,
|
|
4335
|
+
"schema": SEMANTIC_GC_PLAN_SCHEMA_VERSION,
|
|
4336
|
+
"status": "ready_for_plan_review" if not blockers else "blocked",
|
|
4337
|
+
"structurally_valid_unit_count": sum(not row["structural_issues"] for row in rows),
|
|
4338
|
+
"total_unit_count": total_count,
|
|
4339
|
+
"unit_validation": rows,
|
|
4340
|
+
"unreachable_unit_count": unreachable_count,
|
|
4341
|
+
"verification_scope": verification_scope,
|
|
4342
|
+
"warnings": warnings,
|
|
4343
|
+
}
|
|
4344
|
+
|
|
4345
|
+
|
|
4346
|
+
def command_plan_semantic_gc(args: argparse.Namespace) -> int:
|
|
4347
|
+
payload = semantic_gc_plan_payload(args)
|
|
1988
4348
|
if args.json:
|
|
1989
|
-
|
|
4349
|
+
print(json.dumps(payload, ensure_ascii=False, sort_keys=True, separators=(",", ":")))
|
|
1990
4350
|
else:
|
|
1991
|
-
print("ContextGuard
|
|
1992
|
-
print("No image rendering, OCR/image service, model call, proxy forwarding, binary artifact, or replacement was emitted.")
|
|
4351
|
+
print("ContextGuard semantic-gc plan")
|
|
1993
4352
|
print(f"Status: {payload['status']}")
|
|
1994
|
-
|
|
1995
|
-
|
|
1996
|
-
|
|
1997
|
-
|
|
4353
|
+
print(f"Graph complete: {payload['graph_integrity_complete']}; candidates: {payload['candidate_count']}")
|
|
4354
|
+
if payload["blockers"]:
|
|
4355
|
+
print(f"Blockers: {', '.join(payload['blockers'])}")
|
|
4356
|
+
print(f"Exit contract: {payload['process_exit_contract']}.")
|
|
4357
|
+
print("semantic-gc is plan-only; no context was deleted, omitted, read, replaced, or authorized for runtime action.")
|
|
4358
|
+
return 0 if payload["status"] == "ready_for_plan_review" else 2
|
|
1998
4359
|
|
|
1999
4360
|
|
|
2000
|
-
|
|
2001
|
-
|
|
2002
|
-
|
|
2003
|
-
|
|
2004
|
-
|
|
2005
|
-
|
|
2006
|
-
|
|
2007
|
-
|
|
4361
|
+
_STATIC_RELEVANCE_NONFINITE_SENTINEL = object()
|
|
4362
|
+
|
|
4363
|
+
|
|
4364
|
+
def ordered_static_relevance_taxonomy(values: list[str] | set[str]) -> list[str]:
|
|
4365
|
+
selected = set(values)
|
|
4366
|
+
return [value for value in STATIC_RELEVANCE_BLOCKER_ORDER if value in selected]
|
|
4367
|
+
|
|
4368
|
+
|
|
4369
|
+
def static_relevance_validation_row(index: int, issue: str | None = None) -> dict[str, Any]:
|
|
4370
|
+
return {
|
|
4371
|
+
"input_index": index,
|
|
4372
|
+
"unit_id": None,
|
|
4373
|
+
"normalized_path": None,
|
|
4374
|
+
"normalized_evidence_included": False,
|
|
4375
|
+
"structural_issues": [issue] if issue else [],
|
|
4376
|
+
"missing_signals": [],
|
|
4377
|
+
"invalid_signals": [],
|
|
4378
|
+
"protection_reasons": [],
|
|
2008
4379
|
}
|
|
2009
|
-
protected_policy = (args.protected_zone_policy or "deny").strip().lower()
|
|
2010
|
-
missed_context_notes = clean_values(args.missed_context_note)
|
|
2011
|
-
missing_provenance_notes = clean_values(args.missing_provenance_note)
|
|
2012
4380
|
|
|
4381
|
+
|
|
4382
|
+
def decode_static_relevance_unit(raw: Any, index: int) -> tuple[dict[str, Any] | None, dict[str, Any]]:
|
|
4383
|
+
row = static_relevance_validation_row(index)
|
|
4384
|
+
if not isinstance(raw, str):
|
|
4385
|
+
row["structural_issues"] = ["malformed_relevance_unit_json"]
|
|
4386
|
+
return None, row
|
|
4387
|
+
try:
|
|
4388
|
+
encoded = raw.encode("utf-8", errors="strict")
|
|
4389
|
+
except UnicodeEncodeError:
|
|
4390
|
+
row["structural_issues"] = ["invalid_unicode_scalar"]
|
|
4391
|
+
return None, row
|
|
4392
|
+
if len(encoded) > STATIC_RELEVANCE_UNIT_JSON_BYTE_CAP:
|
|
4393
|
+
row["structural_issues"] = ["relevance_unit_json_too_large"]
|
|
4394
|
+
return None, row
|
|
4395
|
+
|
|
4396
|
+
duplicate_key = False
|
|
4397
|
+
|
|
4398
|
+
def pairs_hook(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
|
|
4399
|
+
nonlocal duplicate_key
|
|
4400
|
+
result: dict[str, Any] = {}
|
|
4401
|
+
for key, value in pairs:
|
|
4402
|
+
if key in result:
|
|
4403
|
+
duplicate_key = True
|
|
4404
|
+
result[key] = value
|
|
4405
|
+
return result
|
|
4406
|
+
|
|
4407
|
+
try:
|
|
4408
|
+
decoded = json.loads(
|
|
4409
|
+
raw,
|
|
4410
|
+
object_pairs_hook=pairs_hook,
|
|
4411
|
+
parse_constant=lambda _value: _STATIC_RELEVANCE_NONFINITE_SENTINEL,
|
|
4412
|
+
)
|
|
4413
|
+
except RecursionError:
|
|
4414
|
+
row["structural_issues"] = ["decoder_recursion_limit"]
|
|
4415
|
+
return None, row
|
|
4416
|
+
except UnicodeDecodeError:
|
|
4417
|
+
row["structural_issues"] = ["invalid_unicode_scalar"]
|
|
4418
|
+
return None, row
|
|
4419
|
+
except (json.JSONDecodeError, ValueError, TypeError):
|
|
4420
|
+
row["structural_issues"] = ["malformed_relevance_unit_json"]
|
|
4421
|
+
return None, row
|
|
4422
|
+
|
|
4423
|
+
depth_exceeded = False
|
|
4424
|
+
nonfinite = False
|
|
4425
|
+
invalid_unicode = False
|
|
4426
|
+
try:
|
|
4427
|
+
stack: list[tuple[Any, int]] = [(decoded, 0)]
|
|
4428
|
+
while stack:
|
|
4429
|
+
value, depth = stack.pop()
|
|
4430
|
+
if depth > STATIC_RELEVANCE_JSON_MAX_DEPTH:
|
|
4431
|
+
depth_exceeded = True
|
|
4432
|
+
continue
|
|
4433
|
+
if isinstance(value, str):
|
|
4434
|
+
try:
|
|
4435
|
+
value.encode("utf-8", errors="strict")
|
|
4436
|
+
except UnicodeEncodeError:
|
|
4437
|
+
invalid_unicode = True
|
|
4438
|
+
elif value is _STATIC_RELEVANCE_NONFINITE_SENTINEL or (
|
|
4439
|
+
type(value) is float and not math.isfinite(value)
|
|
4440
|
+
):
|
|
4441
|
+
nonfinite = True
|
|
4442
|
+
if isinstance(value, dict):
|
|
4443
|
+
for key, child in value.items():
|
|
4444
|
+
stack.append((key, depth + 1))
|
|
4445
|
+
stack.append((child, depth + 1))
|
|
4446
|
+
elif isinstance(value, list):
|
|
4447
|
+
for child in value:
|
|
4448
|
+
stack.append((child, depth + 1))
|
|
4449
|
+
except RecursionError:
|
|
4450
|
+
row["structural_issues"] = ["decoder_recursion_limit"]
|
|
4451
|
+
return None, row
|
|
4452
|
+
|
|
4453
|
+
issue = None
|
|
4454
|
+
if duplicate_key:
|
|
4455
|
+
issue = "duplicate_relevance_unit_json_key"
|
|
4456
|
+
elif nonfinite:
|
|
4457
|
+
issue = "non_finite_relevance_unit_json_value"
|
|
4458
|
+
elif depth_exceeded:
|
|
4459
|
+
issue = "relevance_unit_json_depth_exceeded"
|
|
4460
|
+
elif invalid_unicode:
|
|
4461
|
+
issue = "invalid_unicode_scalar"
|
|
4462
|
+
elif not isinstance(decoded, dict):
|
|
4463
|
+
issue = "relevance_unit_must_be_object"
|
|
4464
|
+
if issue:
|
|
4465
|
+
row["structural_issues"] = [issue]
|
|
4466
|
+
return None, row
|
|
4467
|
+
return decoded, row
|
|
4468
|
+
|
|
4469
|
+
|
|
4470
|
+
def valid_static_relevance_path(value: Any) -> bool:
|
|
4471
|
+
if not isinstance(value, str):
|
|
4472
|
+
return False
|
|
4473
|
+
try:
|
|
4474
|
+
encoded = value.encode("utf-8", errors="strict")
|
|
4475
|
+
except UnicodeEncodeError:
|
|
4476
|
+
return False
|
|
4477
|
+
if not (1 <= len(encoded) <= 240) or value.startswith("/") or value.endswith("/") or "\\" in value:
|
|
4478
|
+
return False
|
|
4479
|
+
components = value.split("/")
|
|
4480
|
+
if any(component in {"", ".", ".."} for component in components):
|
|
4481
|
+
return False
|
|
4482
|
+
return not any(unicodedata.category(char) in {"Cc", "Cs"} for char in value)
|
|
4483
|
+
|
|
4484
|
+
|
|
4485
|
+
def valid_static_relevance_symbol_name(value: Any) -> bool:
|
|
4486
|
+
if not isinstance(value, str):
|
|
4487
|
+
return False
|
|
4488
|
+
try:
|
|
4489
|
+
size = len(value.encode("utf-8", errors="strict"))
|
|
4490
|
+
except UnicodeEncodeError:
|
|
4491
|
+
return False
|
|
4492
|
+
return 1 <= size <= 128 and all(
|
|
4493
|
+
char.isprintable() and not unicodedata.category(char).startswith("C") for char in value
|
|
4494
|
+
)
|
|
4495
|
+
|
|
4496
|
+
|
|
4497
|
+
def static_relevance_builtin_protection_reasons(path: str) -> list[str]:
|
|
4498
|
+
reasons: set[str] = set()
|
|
4499
|
+
components = path.split("/")
|
|
4500
|
+
lowered = [component.lower() for component in components]
|
|
4501
|
+
for component in lowered:
|
|
4502
|
+
tokens = {token for token in STATIC_RELEVANCE_PATH_TOKEN_SPLIT_RE.split(component) if token}
|
|
4503
|
+
if tokens & {"auth", "authentication", "authorization"}:
|
|
4504
|
+
reasons.add("builtin_auth_path")
|
|
4505
|
+
if "security" in tokens:
|
|
4506
|
+
reasons.add("builtin_security_path")
|
|
4507
|
+
if tokens & {"secret", "secrets", "credential", "credentials"}:
|
|
4508
|
+
reasons.add("builtin_secret_path")
|
|
4509
|
+
if tokens & {"migration", "migrations"}:
|
|
4510
|
+
reasons.add("builtin_migration_path")
|
|
4511
|
+
if tokens & {"acceptance", "e2e"}:
|
|
4512
|
+
reasons.add("builtin_acceptance_path")
|
|
4513
|
+
basename = lowered[-1]
|
|
4514
|
+
if (
|
|
4515
|
+
basename == ".env"
|
|
4516
|
+
or basename.startswith(".env.")
|
|
4517
|
+
or basename.endswith((".pem", ".key", ".p12", ".pfx"))
|
|
4518
|
+
):
|
|
4519
|
+
reasons.add("builtin_secret_material_path")
|
|
4520
|
+
return [reason for reason in STATIC_RELEVANCE_PROTECTION_REASON_ORDER if reason in reasons]
|
|
4521
|
+
|
|
4522
|
+
|
|
4523
|
+
def normalize_static_relevance_unit(
|
|
4524
|
+
obj: dict[str, Any], row: dict[str, Any]
|
|
4525
|
+
) -> dict[str, Any]:
|
|
4526
|
+
structural: list[str] = []
|
|
4527
|
+
missing: list[str] = []
|
|
4528
|
+
invalid: list[str] = []
|
|
4529
|
+
if obj.get("schema") != STATIC_RELEVANCE_UNIT_SCHEMA_VERSION:
|
|
4530
|
+
structural.append("relevance_unit_schema_mismatch")
|
|
4531
|
+
if set(obj) - STATIC_RELEVANCE_ALLOWED_FIELDS:
|
|
4532
|
+
structural.append("relevance_unit_unexpected_field")
|
|
4533
|
+
|
|
4534
|
+
raw_id = obj.get("unit_id")
|
|
4535
|
+
unit_id = raw_id if isinstance(raw_id, str) and STATIC_RELEVANCE_ID_RE.fullmatch(raw_id) else None
|
|
4536
|
+
if unit_id is None:
|
|
4537
|
+
structural.append("invalid_relevance_unit_id")
|
|
4538
|
+
row["unit_id"] = unit_id
|
|
4539
|
+
|
|
4540
|
+
raw_path = obj.get("path")
|
|
4541
|
+
normalized_path = raw_path if valid_static_relevance_path(raw_path) else None
|
|
4542
|
+
if normalized_path is None:
|
|
4543
|
+
structural.append("invalid_relevance_unit_path")
|
|
4544
|
+
row["normalized_path"] = normalized_path
|
|
4545
|
+
|
|
4546
|
+
task_anchor = obj.get("task_anchor")
|
|
4547
|
+
if type(task_anchor) is not bool:
|
|
4548
|
+
structural.append("invalid_task_anchor")
|
|
4549
|
+
task_anchor = None
|
|
4550
|
+
|
|
4551
|
+
explicit_reasons: list[str] = []
|
|
4552
|
+
if "protection_reasons" not in obj:
|
|
4553
|
+
structural.append("missing_protection_reasons")
|
|
4554
|
+
else:
|
|
4555
|
+
raw_reasons = obj.get("protection_reasons")
|
|
4556
|
+
if (
|
|
4557
|
+
not isinstance(raw_reasons, list)
|
|
4558
|
+
or len(raw_reasons) > 8
|
|
4559
|
+
or any(not isinstance(reason, str) or reason not in STATIC_RELEVANCE_EXPLICIT_PROTECTION_REASONS for reason in raw_reasons)
|
|
4560
|
+
or len(raw_reasons) != len(set(raw_reasons))
|
|
4561
|
+
):
|
|
4562
|
+
structural.append("invalid_protection_reasons")
|
|
4563
|
+
else:
|
|
4564
|
+
explicit_reasons = list(raw_reasons)
|
|
4565
|
+
all_reasons = set(explicit_reasons)
|
|
4566
|
+
if normalized_path is not None:
|
|
4567
|
+
all_reasons.update(static_relevance_builtin_protection_reasons(normalized_path))
|
|
4568
|
+
ordered_reasons = [reason for reason in STATIC_RELEVANCE_PROTECTION_REASON_ORDER if reason in all_reasons]
|
|
4569
|
+
row["protection_reasons"] = ordered_reasons
|
|
4570
|
+
|
|
4571
|
+
symbol = None
|
|
4572
|
+
if "symbol" not in obj:
|
|
4573
|
+
missing.append("missing_symbol_signal")
|
|
4574
|
+
else:
|
|
4575
|
+
raw_symbol = obj.get("symbol")
|
|
4576
|
+
if not isinstance(raw_symbol, dict) or set(raw_symbol) != {"name", "kind", "start_line", "end_line"}:
|
|
4577
|
+
invalid.append("invalid_symbol_signal")
|
|
4578
|
+
else:
|
|
4579
|
+
name = raw_symbol.get("name")
|
|
4580
|
+
kind = raw_symbol.get("kind")
|
|
4581
|
+
start_line = raw_symbol.get("start_line")
|
|
4582
|
+
end_line = raw_symbol.get("end_line")
|
|
4583
|
+
if (
|
|
4584
|
+
not valid_static_relevance_symbol_name(name)
|
|
4585
|
+
or not isinstance(kind, str)
|
|
4586
|
+
or kind not in STATIC_RELEVANCE_SYMBOL_KINDS
|
|
4587
|
+
or type(start_line) is not int
|
|
4588
|
+
or type(end_line) is not int
|
|
4589
|
+
or not (1 <= start_line <= end_line <= 10_000_000)
|
|
4590
|
+
):
|
|
4591
|
+
invalid.append("invalid_symbol_signal")
|
|
4592
|
+
else:
|
|
4593
|
+
symbol = {"name": name, "kind": kind, "start_line": start_line, "end_line": end_line}
|
|
4594
|
+
|
|
4595
|
+
relations: dict[str, list[str] | None] = {}
|
|
4596
|
+
relation_specs = (
|
|
4597
|
+
("symbol_references", "missing_symbol_references_signal", "invalid_symbol_references_signal"),
|
|
4598
|
+
("dataflow_predecessors", "missing_dataflow_predecessors_signal", "invalid_dataflow_predecessors_signal"),
|
|
4599
|
+
("dataflow_successors", "missing_dataflow_successors_signal", "invalid_dataflow_successors_signal"),
|
|
4600
|
+
)
|
|
4601
|
+
for field, missing_token, invalid_token in relation_specs:
|
|
4602
|
+
if field not in obj:
|
|
4603
|
+
missing.append(missing_token)
|
|
4604
|
+
relations[field] = None
|
|
4605
|
+
continue
|
|
4606
|
+
raw_relations = obj.get(field)
|
|
4607
|
+
valid_list = isinstance(raw_relations, list)
|
|
4608
|
+
safe_targets = [
|
|
4609
|
+
target for target in raw_relations
|
|
4610
|
+
if isinstance(target, str) and STATIC_RELEVANCE_ID_RE.fullmatch(target) is not None
|
|
4611
|
+
] if valid_list else []
|
|
4612
|
+
valid_targets = valid_list and len(safe_targets) == len(raw_relations)
|
|
4613
|
+
duplicate_targets = len(safe_targets) != len(set(safe_targets))
|
|
4614
|
+
if not valid_list or len(raw_relations) > 64 or not valid_targets or duplicate_targets:
|
|
4615
|
+
invalid.append(invalid_token)
|
|
4616
|
+
relations[field] = None
|
|
4617
|
+
if duplicate_targets:
|
|
4618
|
+
structural.append("duplicate_relation_target")
|
|
4619
|
+
continue
|
|
4620
|
+
relations[field] = sorted(raw_relations)
|
|
4621
|
+
|
|
4622
|
+
git = None
|
|
4623
|
+
if "git" not in obj:
|
|
4624
|
+
missing.append("missing_git_signal")
|
|
4625
|
+
else:
|
|
4626
|
+
raw_git = obj.get("git")
|
|
4627
|
+
expected_git_fields = {"blame_age_days", "blame_contributor_count", "path_change_count_90d"}
|
|
4628
|
+
if not isinstance(raw_git, dict) or set(raw_git) - expected_git_fields:
|
|
4629
|
+
invalid.append("invalid_git_signal")
|
|
4630
|
+
else:
|
|
4631
|
+
values: dict[str, int] = {}
|
|
4632
|
+
git_specs = (
|
|
4633
|
+
("blame_age_days", "missing_blame_age_signal", "invalid_blame_age_signal", 0, 365000),
|
|
4634
|
+
("blame_contributor_count", "missing_blame_contributor_signal", "invalid_blame_contributor_signal", 1, 10000),
|
|
4635
|
+
("path_change_count_90d", "missing_path_change_count_signal", "invalid_path_change_count_signal", 0, 100000),
|
|
4636
|
+
)
|
|
4637
|
+
for field, missing_token, invalid_token, minimum, maximum in git_specs:
|
|
4638
|
+
if field not in raw_git:
|
|
4639
|
+
missing.append(missing_token)
|
|
4640
|
+
else:
|
|
4641
|
+
value = raw_git[field]
|
|
4642
|
+
if type(value) is not int or not minimum <= value <= maximum:
|
|
4643
|
+
invalid.append(invalid_token)
|
|
4644
|
+
else:
|
|
4645
|
+
values[field] = value
|
|
4646
|
+
if len(values) == 3 and not any(token == "invalid_git_signal" for token in invalid):
|
|
4647
|
+
git = {
|
|
4648
|
+
"blame_age_days": values["blame_age_days"],
|
|
4649
|
+
"blame_contributor_count": values["blame_contributor_count"],
|
|
4650
|
+
"path_change_count_90d": values["path_change_count_90d"],
|
|
4651
|
+
}
|
|
4652
|
+
|
|
4653
|
+
row["structural_issues"] = ordered_static_relevance_taxonomy(structural)
|
|
4654
|
+
row["missing_signals"] = ordered_static_relevance_taxonomy(missing)
|
|
4655
|
+
row["invalid_signals"] = ordered_static_relevance_taxonomy(invalid)
|
|
4656
|
+
locally_valid = not structural and not missing and not invalid
|
|
4657
|
+
normalized = None
|
|
4658
|
+
if locally_valid:
|
|
4659
|
+
assert unit_id is not None and normalized_path is not None and task_anchor is not None
|
|
4660
|
+
assert symbol is not None and git is not None
|
|
4661
|
+
assert all(relations[field] is not None for field, _, _ in relation_specs)
|
|
4662
|
+
normalized = {
|
|
4663
|
+
"unit_id": unit_id,
|
|
4664
|
+
"normalized_path": normalized_path,
|
|
4665
|
+
"task_anchor": task_anchor,
|
|
4666
|
+
"protection_reasons": ordered_reasons,
|
|
4667
|
+
"symbol": symbol,
|
|
4668
|
+
"symbol_references": relations["symbol_references"],
|
|
4669
|
+
"dataflow_predecessors": relations["dataflow_predecessors"],
|
|
4670
|
+
"dataflow_successors": relations["dataflow_successors"],
|
|
4671
|
+
"git": git,
|
|
4672
|
+
}
|
|
4673
|
+
row["normalized_evidence_included"] = True
|
|
4674
|
+
return {
|
|
4675
|
+
"row": row,
|
|
4676
|
+
"unit_id": unit_id,
|
|
4677
|
+
"normalized_path": normalized_path,
|
|
4678
|
+
"task_anchor": task_anchor,
|
|
4679
|
+
"relations": relations,
|
|
4680
|
+
"normalized": normalized,
|
|
4681
|
+
}
|
|
4682
|
+
|
|
4683
|
+
|
|
4684
|
+
def traverse_static_relevance(starts: list[str], adjacency: dict[str, set[str]]) -> set[str]:
|
|
4685
|
+
visited: set[str] = set()
|
|
4686
|
+
pending = list(starts)
|
|
4687
|
+
while pending:
|
|
4688
|
+
unit_id = pending.pop()
|
|
4689
|
+
if unit_id in visited:
|
|
4690
|
+
continue
|
|
4691
|
+
visited.add(unit_id)
|
|
4692
|
+
pending.extend(sorted(adjacency[unit_id] - visited, reverse=True))
|
|
4693
|
+
return visited
|
|
4694
|
+
|
|
4695
|
+
|
|
4696
|
+
def static_relevance_plan_payload(args: argparse.Namespace) -> dict[str, Any]:
|
|
4697
|
+
raw_units = getattr(args, "relevance_unit_json", None) or []
|
|
4698
|
+
input_count = len(raw_units)
|
|
4699
|
+
detailed_count = min(input_count, STATIC_RELEVANCE_DETAILED_UNIT_CAP)
|
|
4700
|
+
overflow_count = max(input_count - STATIC_RELEVANCE_DETAILED_UNIT_CAP, 0)
|
|
4701
|
+
rows: list[dict[str, Any]] = []
|
|
4702
|
+
units: list[dict[str, Any]] = []
|
|
2013
4703
|
blockers: list[str] = []
|
|
2014
|
-
if
|
|
2015
|
-
blockers.append("
|
|
2016
|
-
if
|
|
2017
|
-
blockers.append(
|
|
2018
|
-
|
|
2019
|
-
|
|
2020
|
-
|
|
2021
|
-
|
|
2022
|
-
|
|
2023
|
-
|
|
2024
|
-
|
|
2025
|
-
|
|
2026
|
-
|
|
2027
|
-
|
|
4704
|
+
if input_count == 0:
|
|
4705
|
+
blockers.append("no_relevance_units")
|
|
4706
|
+
if overflow_count:
|
|
4707
|
+
blockers.append("relevance_unit_limit_exceeded")
|
|
4708
|
+
for index, raw in enumerate(raw_units[:STATIC_RELEVANCE_DETAILED_UNIT_CAP]):
|
|
4709
|
+
decoded, row = decode_static_relevance_unit(raw, index)
|
|
4710
|
+
rows.append(row)
|
|
4711
|
+
if decoded is not None:
|
|
4712
|
+
units.append(normalize_static_relevance_unit(decoded, row))
|
|
4713
|
+
|
|
4714
|
+
by_id: dict[str, list[dict[str, Any]]] = {}
|
|
4715
|
+
for unit in units:
|
|
4716
|
+
if unit["unit_id"] is not None:
|
|
4717
|
+
by_id.setdefault(unit["unit_id"], []).append(unit)
|
|
4718
|
+
duplicate_ids = {unit_id for unit_id, matches in by_id.items() if len(matches) > 1}
|
|
4719
|
+
if duplicate_ids:
|
|
4720
|
+
blockers.append("duplicate_relevance_unit_id")
|
|
4721
|
+
for unit_id in duplicate_ids:
|
|
4722
|
+
for unit in by_id[unit_id]:
|
|
4723
|
+
unit["row"]["structural_issues"] = ordered_static_relevance_taxonomy(
|
|
4724
|
+
unit["row"]["structural_issues"] + ["duplicate_relevance_unit_id"]
|
|
4725
|
+
)
|
|
4726
|
+
|
|
4727
|
+
relation_fields = ("symbol_references", "dataflow_predecessors", "dataflow_successors")
|
|
4728
|
+
for unit in units:
|
|
4729
|
+
relation_targets = [
|
|
4730
|
+
target
|
|
4731
|
+
for field in relation_fields
|
|
4732
|
+
for target in (unit["relations"].get(field) or [])
|
|
4733
|
+
]
|
|
4734
|
+
cross_issues: list[str] = []
|
|
4735
|
+
if any(target not in by_id for target in relation_targets):
|
|
4736
|
+
cross_issues.append("unknown_relation_target")
|
|
4737
|
+
if any(target in duplicate_ids for target in relation_targets):
|
|
4738
|
+
cross_issues.append("ambiguous_relation_target")
|
|
4739
|
+
if cross_issues:
|
|
4740
|
+
unit["row"]["structural_issues"] = ordered_static_relevance_taxonomy(
|
|
4741
|
+
unit["row"]["structural_issues"] + cross_issues
|
|
4742
|
+
)
|
|
4743
|
+
|
|
4744
|
+
unique_normalized = {
|
|
4745
|
+
unit_id: matches[0]["normalized"]
|
|
4746
|
+
for unit_id, matches in by_id.items()
|
|
4747
|
+
if len(matches) == 1 and matches[0]["normalized"] is not None
|
|
4748
|
+
}
|
|
4749
|
+
inconsistent_ids: set[str] = set()
|
|
4750
|
+
for unit_id, unit in unique_normalized.items():
|
|
4751
|
+
assert unit is not None
|
|
4752
|
+
for successor in unit["dataflow_successors"]:
|
|
4753
|
+
target = unique_normalized.get(successor)
|
|
4754
|
+
if target is not None and unit_id not in target["dataflow_predecessors"]:
|
|
4755
|
+
inconsistent_ids.add(unit_id)
|
|
4756
|
+
for predecessor in unit["dataflow_predecessors"]:
|
|
4757
|
+
target = unique_normalized.get(predecessor)
|
|
4758
|
+
if target is not None and unit_id not in target["dataflow_successors"]:
|
|
4759
|
+
inconsistent_ids.add(unit_id)
|
|
4760
|
+
if inconsistent_ids:
|
|
4761
|
+
blockers.append("inconsistent_dataflow_relation")
|
|
4762
|
+
for unit_id in inconsistent_ids:
|
|
4763
|
+
for unit in by_id[unit_id]:
|
|
4764
|
+
unit["row"]["structural_issues"] = ordered_static_relevance_taxonomy(
|
|
4765
|
+
unit["row"]["structural_issues"] + ["inconsistent_dataflow_relation"]
|
|
4766
|
+
)
|
|
4767
|
+
|
|
4768
|
+
safe_anchor_ids = sorted({
|
|
4769
|
+
unit["unit_id"] for unit in units
|
|
4770
|
+
if unit["unit_id"] is not None and unit["task_anchor"] is True
|
|
4771
|
+
})
|
|
4772
|
+
if not safe_anchor_ids:
|
|
4773
|
+
blockers.append("no_task_anchor")
|
|
4774
|
+
for row in rows:
|
|
4775
|
+
blockers.extend(row["structural_issues"])
|
|
4776
|
+
blockers.extend(row["missing_signals"])
|
|
4777
|
+
blockers.extend(row["invalid_signals"])
|
|
4778
|
+
|
|
4779
|
+
protected_policy = "deny" if getattr(args, "protected_path_policy", None) == "deny" else None
|
|
4780
|
+
if protected_policy is None:
|
|
4781
|
+
blockers.append("protected_path_policy_required")
|
|
4782
|
+
provider_ack = bool(getattr(args, "provider_boundary_ack", False))
|
|
4783
|
+
if not provider_ack:
|
|
4784
|
+
blockers.append("provider_boundary_ack_required")
|
|
4785
|
+
blockers = ordered_static_relevance_taxonomy(blockers)
|
|
4786
|
+
|
|
4787
|
+
structural_tokens = set(STATIC_RELEVANCE_BLOCKER_ORDER[:19]) | set(STATIC_RELEVANCE_BLOCKER_ORDER[35:39])
|
|
4788
|
+
structural_integrity_complete = not any(token in structural_tokens for token in blockers)
|
|
4789
|
+
unassessable_tokens = set(STATIC_RELEVANCE_BLOCKER_ORDER[2:12])
|
|
4790
|
+
signal_tokens = set(STATIC_RELEVANCE_BLOCKER_ORDER[19:35])
|
|
4791
|
+
declared_signal_fields_complete = (
|
|
4792
|
+
1 <= input_count <= STATIC_RELEVANCE_DETAILED_UNIT_CAP
|
|
4793
|
+
and not any(token in unassessable_tokens or token in signal_tokens for token in blockers)
|
|
4794
|
+
)
|
|
4795
|
+
compilation_performed = (
|
|
4796
|
+
structural_integrity_complete
|
|
4797
|
+
and declared_signal_fields_complete
|
|
4798
|
+
and protected_policy == "deny"
|
|
4799
|
+
and provider_ack
|
|
4800
|
+
)
|
|
2028
4801
|
|
|
4802
|
+
normalized_units = [unit["normalized"] for unit in units if unit["normalized"] is not None]
|
|
4803
|
+
normalized_units.sort(key=lambda unit: (
|
|
4804
|
+
unit["unit_id"], unit["normalized_path"],
|
|
4805
|
+
json.dumps(unit, ensure_ascii=False, sort_keys=True, separators=(",", ":")),
|
|
4806
|
+
))
|
|
4807
|
+
protected_vetoes = [
|
|
4808
|
+
{
|
|
4809
|
+
"unit_id": unit["unit_id"],
|
|
4810
|
+
"normalized_path": unit["normalized_path"],
|
|
4811
|
+
"protection_reasons": unit["row"]["protection_reasons"],
|
|
4812
|
+
"protected_retention_veto": True,
|
|
4813
|
+
"review_priority_tier": 0,
|
|
4814
|
+
}
|
|
4815
|
+
for unit in units
|
|
4816
|
+
if unit["unit_id"] is not None
|
|
4817
|
+
and unit["normalized_path"] is not None
|
|
4818
|
+
and unit["row"]["protection_reasons"]
|
|
4819
|
+
]
|
|
4820
|
+
protected_vetoes.sort(key=lambda row: (
|
|
4821
|
+
row["unit_id"], row["normalized_path"],
|
|
4822
|
+
json.dumps(row["protection_reasons"], ensure_ascii=False, separators=(",", ":")),
|
|
4823
|
+
))
|
|
4824
|
+
|
|
4825
|
+
backward_ids: list[str] = []
|
|
4826
|
+
forward_ids: list[str] = []
|
|
4827
|
+
symbol_ids: list[str] = []
|
|
4828
|
+
review_order: list[dict[str, Any]] = []
|
|
4829
|
+
if compilation_performed:
|
|
4830
|
+
compiled_by_id = {unit["unit_id"]: unit for unit in normalized_units}
|
|
4831
|
+
identifiers = set(compiled_by_id)
|
|
4832
|
+
successors = {unit_id: set(unit["dataflow_successors"]) for unit_id, unit in compiled_by_id.items()}
|
|
4833
|
+
predecessors = {unit_id: set(unit["dataflow_predecessors"]) for unit_id, unit in compiled_by_id.items()}
|
|
4834
|
+
symbol_out = {unit_id: set(unit["symbol_references"]) for unit_id, unit in compiled_by_id.items()}
|
|
4835
|
+
symbol_in = {unit_id: set() for unit_id in identifiers}
|
|
4836
|
+
symbol_graph = {unit_id: set() for unit_id in identifiers}
|
|
4837
|
+
union_graph = {unit_id: set() for unit_id in identifiers}
|
|
4838
|
+
for unit_id in identifiers:
|
|
4839
|
+
for target in symbol_out[unit_id]:
|
|
4840
|
+
symbol_in[target].add(unit_id)
|
|
4841
|
+
symbol_graph[unit_id].add(target)
|
|
4842
|
+
symbol_graph[target].add(unit_id)
|
|
4843
|
+
union_graph[unit_id].add(target)
|
|
4844
|
+
union_graph[target].add(unit_id)
|
|
4845
|
+
for target in successors[unit_id] | predecessors[unit_id]:
|
|
4846
|
+
union_graph[unit_id].add(target)
|
|
4847
|
+
union_graph[target].add(unit_id)
|
|
4848
|
+
anchors = sorted(unit_id for unit_id, unit in compiled_by_id.items() if unit["task_anchor"])
|
|
4849
|
+
forward = traverse_static_relevance(anchors, successors)
|
|
4850
|
+
backward = traverse_static_relevance(anchors, predecessors)
|
|
4851
|
+
symbol_slice = traverse_static_relevance(anchors, symbol_graph)
|
|
4852
|
+
forward_ids = sorted(forward)
|
|
4853
|
+
backward_ids = sorted(backward)
|
|
4854
|
+
symbol_ids = sorted(symbol_slice)
|
|
4855
|
+
distances = {unit_id: 65 for unit_id in identifiers}
|
|
4856
|
+
pending = [(anchor, 0) for anchor in anchors]
|
|
4857
|
+
cursor = 0
|
|
4858
|
+
while cursor < len(pending):
|
|
4859
|
+
unit_id, distance = pending[cursor]
|
|
4860
|
+
cursor += 1
|
|
4861
|
+
if distance >= distances[unit_id]:
|
|
4862
|
+
continue
|
|
4863
|
+
distances[unit_id] = distance
|
|
4864
|
+
for target in sorted(union_graph[unit_id]):
|
|
4865
|
+
if distance + 1 < distances[target]:
|
|
4866
|
+
pending.append((target, distance + 1))
|
|
4867
|
+
rows_for_sort: list[tuple[tuple[Any, ...], dict[str, Any]]] = []
|
|
4868
|
+
for unit_id, unit in compiled_by_id.items():
|
|
4869
|
+
symbol_in_degree = len(symbol_in[unit_id] - {unit_id})
|
|
4870
|
+
symbol_out_degree = len(symbol_out[unit_id] - {unit_id})
|
|
4871
|
+
dataflow_in_degree = len(predecessors[unit_id] - {unit_id})
|
|
4872
|
+
dataflow_out_degree = len(successors[unit_id] - {unit_id})
|
|
4873
|
+
centrality_total = symbol_in_degree + symbol_out_degree + dataflow_in_degree + dataflow_out_degree
|
|
4874
|
+
protected = bool(unit["protection_reasons"])
|
|
4875
|
+
if protected:
|
|
4876
|
+
tier = 0
|
|
4877
|
+
elif unit["task_anchor"]:
|
|
4878
|
+
tier = 1
|
|
4879
|
+
elif unit_id in forward or unit_id in backward:
|
|
4880
|
+
tier = 2
|
|
4881
|
+
elif unit_id in symbol_slice:
|
|
4882
|
+
tier = 3
|
|
4883
|
+
elif distances[unit_id] != 65:
|
|
4884
|
+
tier = 4
|
|
4885
|
+
else:
|
|
4886
|
+
tier = 5
|
|
4887
|
+
rank_key = (
|
|
4888
|
+
tier, distances[unit_id], -centrality_total,
|
|
4889
|
+
-unit["git"]["path_change_count_90d"], unit["git"]["blame_age_days"],
|
|
4890
|
+
unit["normalized_path"], unit_id,
|
|
4891
|
+
)
|
|
4892
|
+
review = {
|
|
4893
|
+
"rank": 0,
|
|
4894
|
+
"unit_id": unit_id,
|
|
4895
|
+
"normalized_path": unit["normalized_path"],
|
|
4896
|
+
"symbol": unit["symbol"],
|
|
4897
|
+
"task_anchor": unit["task_anchor"],
|
|
4898
|
+
"protection_reasons": unit["protection_reasons"],
|
|
4899
|
+
"protected_retention_veto": protected,
|
|
4900
|
+
"review_priority_tier": tier,
|
|
4901
|
+
"task_distance": distances[unit_id],
|
|
4902
|
+
"centrality": {
|
|
4903
|
+
"symbol_in_degree": symbol_in_degree,
|
|
4904
|
+
"symbol_out_degree": symbol_out_degree,
|
|
4905
|
+
"dataflow_in_degree": dataflow_in_degree,
|
|
4906
|
+
"dataflow_out_degree": dataflow_out_degree,
|
|
4907
|
+
"centrality_total": centrality_total,
|
|
4908
|
+
},
|
|
4909
|
+
"git": unit["git"],
|
|
4910
|
+
"rank_key": list(rank_key),
|
|
4911
|
+
}
|
|
4912
|
+
rows_for_sort.append((rank_key, review))
|
|
4913
|
+
rows_for_sort.sort(key=lambda item: item[0])
|
|
4914
|
+
review_order = [row for _, row in rows_for_sort]
|
|
4915
|
+
for rank, row in enumerate(review_order, 1):
|
|
4916
|
+
row["rank"] = rank
|
|
4917
|
+
|
|
4918
|
+
rows.sort(key=lambda row: (
|
|
4919
|
+
row["unit_id"] is None, row["unit_id"] or "",
|
|
4920
|
+
row["normalized_path"] is None, row["normalized_path"] or "", row["input_index"],
|
|
4921
|
+
))
|
|
2029
4922
|
return {
|
|
2030
|
-
"
|
|
2031
|
-
"
|
|
2032
|
-
"
|
|
2033
|
-
"
|
|
2034
|
-
"
|
|
2035
|
-
"
|
|
2036
|
-
"
|
|
2037
|
-
|
|
2038
|
-
|
|
2039
|
-
"
|
|
2040
|
-
"serve_command_available": False,
|
|
2041
|
-
"runtime_behavior_changed": False,
|
|
2042
|
-
"replacement_context_emitted": False,
|
|
2043
|
-
},
|
|
2044
|
-
"external_services": {
|
|
2045
|
-
"called": False,
|
|
2046
|
-
"network": False,
|
|
2047
|
-
"model_calls": False,
|
|
2048
|
-
"provider_calls": False,
|
|
2049
|
-
"proxy_forwarding": False,
|
|
2050
|
-
},
|
|
2051
|
-
"runtime_side_effects": {
|
|
2052
|
-
"files_written": False,
|
|
2053
|
-
"transcript_edited": False,
|
|
2054
|
-
"prompt_edited": False,
|
|
2055
|
-
"context_replaced": False,
|
|
2056
|
-
"stable_runtime_behavior_changed": False,
|
|
2057
|
-
},
|
|
2058
|
-
"checkpoint_metadata": {
|
|
2059
|
-
"goal": goal,
|
|
2060
|
-
"constraints": clean_values(args.constraint),
|
|
2061
|
-
"decisions": clean_values(args.decision),
|
|
2062
|
-
"open_tasks": clean_values(args.open_task),
|
|
2063
|
-
"evidence_provenance_handles": clean_values(args.evidence_handle),
|
|
2064
|
-
"unresolved_questions": clean_values(args.unresolved_question),
|
|
2065
|
-
},
|
|
2066
|
-
"exact_context_fallback": {
|
|
2067
|
-
"required": True,
|
|
2068
|
-
"available": bool(reexpand_valid),
|
|
2069
|
-
"receipt_id": receipt_id,
|
|
2070
|
-
"reexpand_command": reexpand_command,
|
|
2071
|
-
"verified": False,
|
|
2072
|
-
"must_be_verified_before_checkpoint_metadata_is_used": True,
|
|
2073
|
-
"allowed_reexpand_shapes": [
|
|
2074
|
-
"context-guard-artifact get RECEIPT --full",
|
|
2075
|
-
"context-guard artifact get RECEIPT --full",
|
|
2076
|
-
],
|
|
2077
|
-
},
|
|
2078
|
-
"protected_zones": {
|
|
2079
|
-
"policy": protected_policy,
|
|
2080
|
-
"override_allowed": False,
|
|
2081
|
-
"denied_classes": [
|
|
2082
|
-
"code",
|
|
2083
|
-
"diffs",
|
|
2084
|
-
"identifiers",
|
|
2085
|
-
"hashes",
|
|
2086
|
-
"paths",
|
|
2087
|
-
"numeric_constants",
|
|
2088
|
-
"json_keys",
|
|
2089
|
-
"stack_frames",
|
|
2090
|
-
"secrets",
|
|
2091
|
-
"prompt_like_instructions",
|
|
2092
|
-
],
|
|
4923
|
+
"schema": STATIC_RELEVANCE_PLAN_SCHEMA_VERSION,
|
|
4924
|
+
"experiment_id": "static-relevance",
|
|
4925
|
+
"mode": "plan",
|
|
4926
|
+
"status": "ready_for_plan_review" if compilation_performed else "blocked",
|
|
4927
|
+
"process_exit_contract": STATIC_RELEVANCE_PROCESS_EXIT_CONTRACT,
|
|
4928
|
+
"protected_path_policy": protected_policy,
|
|
4929
|
+
"effective_protected_path_policy": "deny",
|
|
4930
|
+
"provider_boundary_acknowledged": provider_ack,
|
|
4931
|
+
"input_summary": {
|
|
4932
|
+
"input_count": input_count, "detailed_count": detailed_count, "overflow_count": overflow_count,
|
|
2093
4933
|
},
|
|
2094
|
-
"
|
|
2095
|
-
|
|
2096
|
-
|
|
2097
|
-
|
|
2098
|
-
|
|
2099
|
-
|
|
2100
|
-
"
|
|
4934
|
+
"structural_integrity_complete": structural_integrity_complete,
|
|
4935
|
+
"declared_signal_fields_complete": declared_signal_fields_complete,
|
|
4936
|
+
"compilation_performed": compilation_performed,
|
|
4937
|
+
"normalized_evidence": {"unit_count": len(normalized_units), "units": normalized_units},
|
|
4938
|
+
"unit_validation": rows,
|
|
4939
|
+
"compilation": {
|
|
4940
|
+
"task_anchor_ids": safe_anchor_ids,
|
|
4941
|
+
"backward_dataflow_slice_ids": backward_ids,
|
|
4942
|
+
"forward_dataflow_slice_ids": forward_ids,
|
|
4943
|
+
"symbol_slice_ids": symbol_ids,
|
|
4944
|
+
"protected_vetoes": protected_vetoes,
|
|
4945
|
+
"review_order": review_order,
|
|
2101
4946
|
},
|
|
2102
|
-
"
|
|
2103
|
-
|
|
2104
|
-
|
|
2105
|
-
"
|
|
2106
|
-
"
|
|
2107
|
-
"
|
|
4947
|
+
"readiness_blockers": blockers,
|
|
4948
|
+
"warnings": list(STATIC_RELEVANCE_WARNING_ORDER),
|
|
4949
|
+
"verification_scope": {
|
|
4950
|
+
"evidence_collection_verified": False,
|
|
4951
|
+
"repository_coverage_verified": False,
|
|
4952
|
+
"symbol_resolution_verified": False,
|
|
4953
|
+
"dataflow_semantics_verified": False,
|
|
4954
|
+
"git_metrics_verified": False,
|
|
2108
4955
|
},
|
|
2109
|
-
"
|
|
2110
|
-
"
|
|
2111
|
-
"
|
|
2112
|
-
"
|
|
2113
|
-
|
|
2114
|
-
|
|
2115
|
-
|
|
2116
|
-
"Measure provider/model token and cost fields on matched successful tasks before any hosted savings claim.",
|
|
2117
|
-
],
|
|
4956
|
+
"runtime_boundaries": {
|
|
4957
|
+
"repository_scanned": False,
|
|
4958
|
+
"source_content_read": False,
|
|
4959
|
+
"git_invoked": False,
|
|
4960
|
+
"parser_invoked": False,
|
|
4961
|
+
"provider_called": False,
|
|
4962
|
+
"files_written": False,
|
|
2118
4963
|
},
|
|
2119
|
-
"claim_boundary": (
|
|
2120
|
-
"Dry-run semantic-checkpoint planning only; checkpoint metadata is not replacement context and no hosted "
|
|
2121
|
-
"token/cost savings claim is allowed without provider-measured matched successful tasks."
|
|
2122
|
-
),
|
|
2123
4964
|
"candidate_replacement": None,
|
|
4965
|
+
"human_review_performed": False,
|
|
4966
|
+
"deprioritization_authorized": False,
|
|
4967
|
+
"omission_authorized": False,
|
|
4968
|
+
"runtime_action_allowed": False,
|
|
2124
4969
|
}
|
|
2125
4970
|
|
|
2126
4971
|
|
|
2127
|
-
def
|
|
2128
|
-
payload =
|
|
4972
|
+
def command_plan_static_relevance(args: argparse.Namespace) -> int:
|
|
4973
|
+
payload = static_relevance_plan_payload(args)
|
|
2129
4974
|
if args.json:
|
|
2130
|
-
|
|
4975
|
+
print(json.dumps(payload, ensure_ascii=False, sort_keys=True, separators=(",", ":")))
|
|
2131
4976
|
else:
|
|
2132
|
-
|
|
2133
|
-
print("
|
|
4977
|
+
summary = payload["input_summary"]
|
|
4978
|
+
print("ContextGuard static-relevance plan")
|
|
2134
4979
|
print(f"Status: {payload['status']}")
|
|
2135
|
-
|
|
2136
|
-
|
|
2137
|
-
print(
|
|
2138
|
-
return 0
|
|
4980
|
+
print(f"Inputs: {summary['input_count']}; blockers: {len(payload['readiness_blockers'])}")
|
|
4981
|
+
print(f"Protected retention vetoes: {len(payload['compilation']['protected_vetoes'])}")
|
|
4982
|
+
print("Static relevance is plan-review-only and authorizes no deprioritization, omission, deletion, or runtime action.")
|
|
4983
|
+
return 0 if payload["status"] == "ready_for_plan_review" else 2
|
|
2139
4984
|
|
|
2140
4985
|
|
|
2141
4986
|
def visual_crop_ocr_evidence_pack_payload(args: argparse.Namespace) -> dict[str, Any]:
|
|
@@ -4838,7 +7683,9 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
4838
7683
|
add_common_args(disable_parser)
|
|
4839
7684
|
disable_parser.set_defaults(func=command_disable)
|
|
4840
7685
|
|
|
4841
|
-
plan_parser = sub.add_parser(
|
|
7686
|
+
plan_parser = sub.add_parser(
|
|
7687
|
+
"plan", allow_abbrev=False, help="Run read-only dry-run planners for experimental lanes."
|
|
7688
|
+
)
|
|
4842
7689
|
plan_sub = plan_parser.add_subparsers(dest="plan_command", required=True)
|
|
4843
7690
|
|
|
4844
7691
|
context_diff = plan_sub.add_parser(
|
|
@@ -4922,6 +7769,106 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
4922
7769
|
semantic_checkpoint.add_argument("--json", action="store_true", help="Emit JSON output.")
|
|
4923
7770
|
semantic_checkpoint.set_defaults(func=command_plan_semantic_checkpoint)
|
|
4924
7771
|
|
|
7772
|
+
proof_carrying_context = plan_sub.add_parser(
|
|
7773
|
+
"proof-carrying-context",
|
|
7774
|
+
help="Dry-run bounded proof-envelope metadata readiness without reading or verifying content.",
|
|
7775
|
+
)
|
|
7776
|
+
proof_carrying_context.add_argument(
|
|
7777
|
+
"--proof-unit-json",
|
|
7778
|
+
action="append",
|
|
7779
|
+
help="Inline literal proof-unit JSON object. Repeatable; never treated as a path.",
|
|
7780
|
+
)
|
|
7781
|
+
proof_carrying_context.add_argument(
|
|
7782
|
+
"--provider-boundary-ack",
|
|
7783
|
+
action="store_true",
|
|
7784
|
+
help="Acknowledge hosted claims require provider-measured matched successful tasks for the target model.",
|
|
7785
|
+
)
|
|
7786
|
+
proof_carrying_context.add_argument(
|
|
7787
|
+
"--protected-zone-policy",
|
|
7788
|
+
default="deny",
|
|
7789
|
+
choices=("deny", "allow"),
|
|
7790
|
+
help="Caller-declared protected evidence policy; only deny can pass and compliance remains unchecked.",
|
|
7791
|
+
)
|
|
7792
|
+
proof_carrying_context.add_argument("--json", action="store_true", help="Emit JSON output.")
|
|
7793
|
+
proof_carrying_context.set_defaults(func=command_plan_proof_carrying_context)
|
|
7794
|
+
|
|
7795
|
+
verify_parser = sub.add_parser(
|
|
7796
|
+
"verify",
|
|
7797
|
+
allow_abbrev=False,
|
|
7798
|
+
help="Run bounded read-only local verifiers for experimental lanes.",
|
|
7799
|
+
)
|
|
7800
|
+
verify_sub = verify_parser.add_subparsers(dest="verify_command", required=True)
|
|
7801
|
+
verify_proof = verify_sub.add_parser(
|
|
7802
|
+
"proof-carrying-context",
|
|
7803
|
+
allow_abbrev=False,
|
|
7804
|
+
help="Verify explicit private receipt leaves without retrieval or execution.",
|
|
7805
|
+
)
|
|
7806
|
+
verify_proof.add_argument(
|
|
7807
|
+
"--artifact-dir",
|
|
7808
|
+
action=StoreOnceAction,
|
|
7809
|
+
help="One explicit private local artifact directory; no fallback is searched.",
|
|
7810
|
+
)
|
|
7811
|
+
verify_proof.add_argument(
|
|
7812
|
+
"--proof-unit-json",
|
|
7813
|
+
action="append",
|
|
7814
|
+
help="Inline literal proof-unit JSON object. Repeatable; never treated as a path.",
|
|
7815
|
+
)
|
|
7816
|
+
verify_proof.add_argument("--json", action="store_true", help="Emit JSON output.")
|
|
7817
|
+
verify_proof.set_defaults(func=command_verify_proof_carrying_context)
|
|
7818
|
+
|
|
7819
|
+
semantic_gc = plan_sub.add_parser(
|
|
7820
|
+
"semantic-gc",
|
|
7821
|
+
allow_abbrev=False,
|
|
7822
|
+
help="Plan caller-declared graph reachability candidates without reading or omitting context.",
|
|
7823
|
+
)
|
|
7824
|
+
semantic_gc.add_argument(
|
|
7825
|
+
"--context-unit-json",
|
|
7826
|
+
action="append",
|
|
7827
|
+
help="Inline literal semantic-GC unit JSON object. Repeatable; never treated as a path.",
|
|
7828
|
+
)
|
|
7829
|
+
semantic_gc.add_argument(
|
|
7830
|
+
"--provider-boundary-ack",
|
|
7831
|
+
action="store_true",
|
|
7832
|
+
help="Acknowledge that provider behavior and hosted savings remain unverified.",
|
|
7833
|
+
)
|
|
7834
|
+
semantic_gc.add_argument(
|
|
7835
|
+
"--human-review-ack",
|
|
7836
|
+
action="store_true",
|
|
7837
|
+
help="Acknowledge that candidate review remains required; this does not perform review.",
|
|
7838
|
+
)
|
|
7839
|
+
semantic_gc.add_argument(
|
|
7840
|
+
"--protected-zone-policy",
|
|
7841
|
+
choices=("deny",),
|
|
7842
|
+
default=None,
|
|
7843
|
+
help="Explicit deny-only protected-zone declaration; omitted remains effective deny but blocks readiness.",
|
|
7844
|
+
)
|
|
7845
|
+
semantic_gc.add_argument("--json", action="store_true", help="Emit JSON output.")
|
|
7846
|
+
semantic_gc.set_defaults(func=command_plan_semantic_gc)
|
|
7847
|
+
|
|
7848
|
+
static_relevance = plan_sub.add_parser(
|
|
7849
|
+
"static-relevance",
|
|
7850
|
+
allow_abbrev=False,
|
|
7851
|
+
help="Compile bounded caller-declared static evidence into plan-review diagnostics.",
|
|
7852
|
+
)
|
|
7853
|
+
static_relevance.add_argument(
|
|
7854
|
+
"--relevance-unit-json",
|
|
7855
|
+
action="append",
|
|
7856
|
+
help="Inline literal static-relevance unit JSON object. Repeatable; never treated as a path.",
|
|
7857
|
+
)
|
|
7858
|
+
static_relevance.add_argument(
|
|
7859
|
+
"--provider-boundary-ack",
|
|
7860
|
+
action="store_true",
|
|
7861
|
+
help="Acknowledge that provider behavior and hosted savings remain unverified.",
|
|
7862
|
+
)
|
|
7863
|
+
static_relevance.add_argument(
|
|
7864
|
+
"--protected-path-policy",
|
|
7865
|
+
choices=("deny",),
|
|
7866
|
+
default=None,
|
|
7867
|
+
help="Explicit deny-only protected-path declaration; omitted remains effective deny but blocks readiness.",
|
|
7868
|
+
)
|
|
7869
|
+
static_relevance.add_argument("--json", action="store_true", help="Emit JSON output.")
|
|
7870
|
+
static_relevance.set_defaults(func=command_plan_static_relevance)
|
|
7871
|
+
|
|
4925
7872
|
self_hosted = plan_sub.add_parser(
|
|
4926
7873
|
"self-hosted-metrics-ledger",
|
|
4927
7874
|
help="Dry-run self-hosted/local metrics ledger sidecar evidence without writing a ledger.",
|