sidegraph 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,400 @@
1
+ """Pure conversion of bounded scan results into immutable Bootstrap candidates."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import json
7
+ import re
8
+ from pathlib import Path
9
+ from typing import TYPE_CHECKING
10
+
11
+ from sidegraph.bootstrap.catalog import CanonicalCatalog, fingerprint_catalog
12
+ from sidegraph.bootstrap.model import (
13
+ AnchorPlan,
14
+ BootstrapCandidate,
15
+ BootstrapPlan,
16
+ EditableCandidate,
17
+ PlanIssue,
18
+ ScanResult,
19
+ SourceFingerprint,
20
+ WarningCode,
21
+ )
22
+ from sidegraph.capture import redact
23
+ from sidegraph.doc_import import (
24
+ ParsedDoc,
25
+ _is_draft_like_status,
26
+ _is_live_tree_path,
27
+ _is_rejected_status,
28
+ parse_decision_docs,
29
+ )
30
+ from sidegraph.profiles import FlowProfile
31
+ from sidegraph.schema import DecisionKind, DecisionStatus, Descriptor
32
+
33
+ if TYPE_CHECKING:
34
+ from sidegraph.engine.reader import GraphifyReader
35
+
36
+
37
+ _CURRENT_STATE_PREFIXES = (
38
+ "currently ",
39
+ "today ",
40
+ "the system is ",
41
+ "the system has ",
42
+ "the system uses ",
43
+ )
44
+ _DECISION_VERBS = ("choose", "decide", "adopt", "require", "must", "will", "use ")
45
+ _MENTION_RE = re.compile(r"`([^`\n]+)`")
46
+
47
+
48
+ def _quality_warnings(parsed: ParsedDoc) -> tuple[WarningCode, ...]:
49
+ warnings: list[WarningCode] = []
50
+ choice = parsed.choice.strip().lower()
51
+ if not parsed.choice.strip():
52
+ warnings.append(WarningCode.MISSING_CHOICE)
53
+ if not (parsed.rejected or "").strip():
54
+ warnings.append(WarningCode.MISSING_REJECTED)
55
+ if choice.startswith(_CURRENT_STATE_PREFIXES) and not any(
56
+ verb in choice for verb in _DECISION_VERBS
57
+ ):
58
+ warnings.append(WarningCode.CURRENT_STATE)
59
+ return tuple(warnings)
60
+
61
+
62
+ def _default_status(parsed: ParsedDoc) -> DecisionStatus:
63
+ if _is_rejected_status(parsed.frontmatter_status):
64
+ return DecisionStatus.REJECTED
65
+ if _is_draft_like_status(parsed.frontmatter_status):
66
+ return DecisionStatus.PROPOSED
67
+ return DecisionStatus.ACCEPTED
68
+
69
+
70
+ def _anchor_text(parsed: ParsedDoc) -> str:
71
+ return "\n".join(
72
+ value
73
+ for value in (
74
+ parsed.title,
75
+ parsed.context,
76
+ parsed.choice,
77
+ parsed.rejected,
78
+ parsed.consequences,
79
+ )
80
+ if value
81
+ )
82
+
83
+
84
+ def _context_body(parsed: ParsedDoc, ref: str) -> str:
85
+ suffix = f"\n\nimported from {ref}"
86
+ return parsed.context.removesuffix(suffix).strip()
87
+
88
+
89
+ def _choice_is_context_fallback(parsed: ParsedDoc, ref: str) -> bool:
90
+ """Reject the parser's last-resort context echo as a Bootstrap decision choice."""
91
+ context = _context_body(parsed, ref)
92
+ return bool(context) and parsed.choice.strip() == context
93
+
94
+
95
+ def _candidate(parsed: ParsedDoc, file_path: str, ref: str, source_hash: str) -> BootstrapCandidate:
96
+ """``file_path`` stays the ON-DISK path (anchor lookups, file I/O); ``ref`` is derived
97
+ by the caller from the NORMALIZED path (E2, design note §6) — one field cannot serve
98
+ both after an archive-style move (review M3)."""
99
+ return BootstrapCandidate.from_fields(
100
+ file_path=file_path,
101
+ ref=ref,
102
+ fragment=parsed.fragment,
103
+ source_hash=source_hash,
104
+ title=parsed.title,
105
+ context=parsed.context,
106
+ choice=parsed.choice,
107
+ rejected=parsed.rejected,
108
+ consequences=parsed.consequences,
109
+ kind=parsed.suggested_kind or DecisionKind.ADR,
110
+ default_status=_default_status(parsed),
111
+ redacted_anchor_text=_anchor_text(parsed),
112
+ anchor_intents=(),
113
+ file_anchor_intent=None,
114
+ anchors=(),
115
+ warnings=_quality_warnings(parsed),
116
+ )
117
+
118
+
119
+ def _content_signature(candidate: BootstrapCandidate) -> str:
120
+ provenance = f"imported from {candidate.ref}"
121
+ context = (
122
+ ""
123
+ if candidate.context == provenance
124
+ else candidate.context.removesuffix(f"\n\n{provenance}")
125
+ )
126
+ material = "\0".join(
127
+ str(value or "")
128
+ for value in (
129
+ candidate.title,
130
+ context,
131
+ candidate.choice,
132
+ candidate.rejected,
133
+ candidate.consequences,
134
+ candidate.kind,
135
+ )
136
+ )
137
+ return hashlib.sha256(material.encode()).hexdigest()
138
+
139
+
140
+ def _plan_fingerprint(
141
+ profile: str,
142
+ sources: list[SourceFingerprint],
143
+ candidates: list[BootstrapCandidate],
144
+ catalog_fingerprint: str,
145
+ graph_fingerprint: str,
146
+ ) -> str:
147
+ material = {
148
+ "profile": profile,
149
+ "sources": [(item.path, item.sha256) for item in sources],
150
+ "candidate_keys": [candidate.key for candidate in candidates],
151
+ "catalog_fingerprint": catalog_fingerprint,
152
+ "graph_fingerprint": graph_fingerprint,
153
+ }
154
+ encoded = json.dumps(material, ensure_ascii=False, separators=(",", ":")).encode()
155
+ return hashlib.sha256(encoded).hexdigest()
156
+
157
+
158
+ _DUPLICATE_STATUSES = (
159
+ DecisionStatus.ACCEPTED,
160
+ DecisionStatus.PROPOSED,
161
+ DecisionStatus.REJECTED,
162
+ )
163
+
164
+
165
+ def _mention_descriptors(text: str) -> tuple[Descriptor, ...]:
166
+ seen: set[str] = set()
167
+ descriptors: list[Descriptor] = []
168
+ for match in _MENTION_RE.finditer(text):
169
+ name = match.group(1).strip()
170
+ if name and name not in seen:
171
+ seen.add(name)
172
+ descriptors.append(Descriptor(name=name))
173
+ return tuple(descriptors)
174
+
175
+
176
+ def _anchor_plan(descriptor: Descriptor, reader: GraphifyReader) -> AnchorPlan:
177
+ result = reader.resolve(descriptor)
178
+ if result.status == "resolved":
179
+ return AnchorPlan(descriptor=descriptor, status="resolved", tier=2)
180
+ if result.status == "ambiguous":
181
+ return AnchorPlan(
182
+ descriptor=descriptor,
183
+ status="ambiguous",
184
+ candidates=tuple(result.candidates),
185
+ tier=1 if result.community is not None else None,
186
+ )
187
+ return AnchorPlan(descriptor=descriptor, status="unresolved", tier=2)
188
+
189
+
190
+ def _enrich_candidate(
191
+ candidate: BootstrapCandidate,
192
+ *,
193
+ catalog: CanonicalCatalog | None,
194
+ reader: GraphifyReader | None,
195
+ ) -> BootstrapCandidate:
196
+ warnings = list(candidate.warnings)
197
+ if catalog is not None and catalog.find_by_ref(
198
+ "doc-import", candidate.ref, _DUPLICATE_STATUSES
199
+ ):
200
+ warnings.append(WarningCode.DUPLICATE_CANONICAL)
201
+
202
+ anchor_intents: tuple[Descriptor, ...] = ()
203
+ file_anchor_intent: Descriptor | None = None
204
+ anchors: tuple[AnchorPlan, ...] = ()
205
+ if reader is not None:
206
+ anchor_intents = _mention_descriptors(candidate.redacted_anchor_text)
207
+ displayed = [_anchor_plan(descriptor, reader) for descriptor in anchor_intents]
208
+
209
+ document_nodes = sorted(
210
+ (
211
+ node
212
+ for node in reader.nodes_in_file(candidate.file_path)
213
+ if node.file_type == "document"
214
+ ),
215
+ key=lambda node: (node.node_id, node.name),
216
+ )
217
+ if document_nodes:
218
+ chosen = document_nodes[0]
219
+ file_anchor_intent = Descriptor(
220
+ name=chosen.name,
221
+ file_path=candidate.file_path,
222
+ )
223
+ displayed.append(_anchor_plan(file_anchor_intent, reader))
224
+
225
+ anchors = tuple(displayed)
226
+ if any(anchor.status == "ambiguous" for anchor in anchors):
227
+ warnings.append(WarningCode.AMBIGUOUS_ANCHOR)
228
+ if any(anchor.status == "unresolved" for anchor in anchors):
229
+ warnings.append(WarningCode.UNRESOLVED_ANCHOR)
230
+
231
+ return candidate.model_copy(
232
+ update={
233
+ "anchor_intents": anchor_intents,
234
+ "file_anchor_intent": file_anchor_intent,
235
+ "anchors": anchors,
236
+ "warnings": tuple(warnings),
237
+ }
238
+ )
239
+
240
+
241
+ def plan_sources(
242
+ root: Path,
243
+ scan: ScanResult,
244
+ profile: FlowProfile,
245
+ *,
246
+ catalog: CanonicalCatalog | None = None,
247
+ reader: GraphifyReader | None = None,
248
+ ) -> BootstrapPlan:
249
+ """Plan redacted candidates without constructing a graph reader or persistent Store."""
250
+ candidates: list[BootstrapCandidate] = []
251
+ issues: list[PlanIssue] = []
252
+ source_fingerprints: list[SourceFingerprint] = []
253
+ files_read: list[str] = []
254
+ content_signatures: set[str] = set()
255
+
256
+ for file_path in scan.files:
257
+ source_bytes = (root / file_path).read_bytes()
258
+ source_hash = hashlib.sha256(source_bytes).hexdigest()
259
+ source_fingerprints.append(SourceFingerprint(path=file_path, sha256=source_hash))
260
+ files_read.append(file_path)
261
+ # E2 (design note §6, review M3): the NORMALIZED path is what gets parsed — that
262
+ # call is what stamps `context`'s "imported from …" and each child's
263
+ # `effective_ref`, so the stamp and fragments come out normalized at the source,
264
+ # matching what `sidegraph-import` stamps for the same doc (§6 point 2). E1: the
265
+ # profile's `title_pattern` rides along, threaded exactly like `dialect`.
266
+ normalized_path = profile.normalized_rel_path(file_path)
267
+ parsed_docs, reason = parse_decision_docs(
268
+ source_bytes.decode("utf-8"),
269
+ normalized_path,
270
+ dialect=profile.dialect,
271
+ title_pattern=profile.title_pattern,
272
+ )
273
+ if not parsed_docs:
274
+ if reason == "unparseable":
275
+ issues.append(
276
+ PlanIssue(
277
+ file_path=file_path,
278
+ ref=normalized_path,
279
+ warning=WarningCode.MISSING_CHOICE,
280
+ detail="Decision-shaped document has no usable choice.",
281
+ )
282
+ )
283
+ continue
284
+
285
+ for parsed in parsed_docs:
286
+ # `ref` is derived from the SAME normalized path the stamp used (E2) — keeps
287
+ # `_context_body`'s `removesuffix` matching, so `_choice_is_context_fallback`
288
+ # (the echo-refusal check) fires exactly like it does pre-E2 (review T15).
289
+ ref = (
290
+ normalized_path
291
+ if parsed.fragment is None
292
+ else f"{normalized_path}#{parsed.fragment}"
293
+ )
294
+ if _choice_is_context_fallback(parsed, ref):
295
+ issues.append(
296
+ PlanIssue(
297
+ file_path=file_path,
298
+ ref=ref,
299
+ warning=WarningCode.MISSING_CHOICE,
300
+ detail="Decision choice only repeats the context fallback.",
301
+ )
302
+ )
303
+ continue
304
+ # I2 (R1 improvement wave §2, Blocker 1 — decide-then-stamp): the in-flight
305
+ # note is appended to `parsed.context` only AFTER the echo-refusal decision
306
+ # above has run against the note-free context — mirrors `import_docs`'s own
307
+ # sequencing (`doc_import._is_live_tree_path` is the shared trigger predicate
308
+ # both write paths use). Stamping any earlier would corrupt
309
+ # `_choice_is_context_fallback`'s own strip-and-compare the same way it would
310
+ # corrupt `_choice_is_context_echo`'s.
311
+ if profile.in_flight_note and _is_live_tree_path(file_path, profile):
312
+ parsed = parsed.model_copy(
313
+ update={"context": f"{parsed.context}\n\n{profile.in_flight_note}"}
314
+ )
315
+ candidate = _candidate(parsed, file_path, ref, source_hash)
316
+ content_signature = _content_signature(candidate)
317
+ if content_signature in content_signatures:
318
+ issues.append(
319
+ PlanIssue(
320
+ file_path=file_path,
321
+ ref=candidate.ref,
322
+ warning=WarningCode.DUPLICATE_PLAN,
323
+ detail=f"Duplicate candidate at {candidate.ref} was omitted.",
324
+ )
325
+ )
326
+ continue
327
+ content_signatures.add(content_signature)
328
+ candidates.append(_enrich_candidate(candidate, catalog=catalog, reader=reader))
329
+
330
+ catalog_digest = fingerprint_catalog(catalog) if catalog is not None else ""
331
+ graph_version = reader.graph_version() if reader is not None else None
332
+ fingerprint = _plan_fingerprint(
333
+ profile.name,
334
+ source_fingerprints,
335
+ candidates,
336
+ catalog_digest,
337
+ graph_version or "",
338
+ )
339
+ return BootstrapPlan(
340
+ root=root.resolve().as_posix(),
341
+ profile=profile.name,
342
+ fingerprint=fingerprint,
343
+ source_fingerprints=tuple(source_fingerprints),
344
+ catalog_fingerprint=catalog_digest,
345
+ graph_version=graph_version,
346
+ candidates=tuple(candidates),
347
+ issues=tuple(issues),
348
+ files_read=tuple(files_read),
349
+ exclusions=scan.exclusions,
350
+ )
351
+
352
+
353
+ def _redacted_edit(editable: EditableCandidate) -> ParsedDoc:
354
+ title, _ = redact(editable.title)
355
+ context, _ = redact(editable.context)
356
+ choice, _ = redact(editable.choice)
357
+ rejected = None
358
+ if editable.rejected is not None:
359
+ rejected, _ = redact(editable.rejected)
360
+ consequences = None
361
+ if editable.consequences is not None:
362
+ consequences, _ = redact(editable.consequences)
363
+ return ParsedDoc(
364
+ title=title,
365
+ context=context,
366
+ choice=choice,
367
+ rejected=rejected,
368
+ consequences=consequences,
369
+ suggested_kind=editable.kind,
370
+ )
371
+
372
+
373
+ def replan_edited_candidate(
374
+ candidate: BootstrapCandidate,
375
+ editable: EditableCandidate,
376
+ *,
377
+ reader: GraphifyReader | None,
378
+ catalog: CanonicalCatalog | None,
379
+ ) -> BootstrapCandidate:
380
+ """Rebuild an edited candidate from redacted fields, resetting unresolved enrichment."""
381
+ parsed = _redacted_edit(editable)
382
+ replanned = BootstrapCandidate.from_fields(
383
+ file_path=candidate.file_path,
384
+ ref=candidate.ref,
385
+ fragment=candidate.fragment,
386
+ source_hash=candidate.source_hash,
387
+ title=parsed.title,
388
+ context=parsed.context,
389
+ choice=parsed.choice,
390
+ rejected=parsed.rejected,
391
+ consequences=parsed.consequences,
392
+ kind=editable.kind,
393
+ default_status=candidate.default_status,
394
+ redacted_anchor_text=_anchor_text(parsed),
395
+ anchor_intents=(),
396
+ file_anchor_intent=None,
397
+ anchors=(),
398
+ warnings=_quality_warnings(parsed),
399
+ )
400
+ return _enrich_candidate(replanned, catalog=catalog, reader=reader)
@@ -0,0 +1,103 @@
1
+ """Read-only proof that a ratified bootstrap record reaches production retrieval."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Collection
6
+ from datetime import UTC, datetime
7
+
8
+ from sidegraph.bootstrap.model import ProofResult, ProofSelection
9
+ from sidegraph.engine.reader import GraphifyReader
10
+ from sidegraph.retrieval import Seed, get_task_context
11
+ from sidegraph.schema import DecisionKind, DecisionStatus, EntityKind
12
+ from sidegraph.store import Store
13
+
14
+ _SELECTION_RULE = (
15
+ "accepted -> valid -> live tier-2 -> gotcha/lesson-or-rejected "
16
+ "-> newest valid_from -> stable id -> lexicographically first non-empty concrete "
17
+ "entity file_path"
18
+ )
19
+
20
+
21
+ def _proof_sort_key(item: ProofSelection) -> tuple[int, float, str]:
22
+ decision = item.decision
23
+ strong = decision.kind in (DecisionKind.GOTCHA, DecisionKind.LESSON) or bool(
24
+ (decision.rejected or "").strip()
25
+ )
26
+ return (0 if strong else 1, -decision.valid_from.timestamp(), decision.id)
27
+
28
+
29
+ def _live_leaf_paths(store: Store, record_id: str) -> list[str]:
30
+ """Concrete file paths from live tier-2 bindings, sorted for a stable seed."""
31
+ paths: set[str] = set()
32
+ for binding in store.bindings_for_record(record_id):
33
+ if binding.tier != 2 or binding.status != "live":
34
+ continue
35
+ entity = store.get_entity(binding.entity_id)
36
+ if entity is None or entity.kind != EntityKind.CONCRETE or entity.descriptor is None:
37
+ continue
38
+ file_path = entity.descriptor.file_path
39
+ if file_path:
40
+ paths.add(file_path)
41
+ return sorted(paths)
42
+
43
+
44
+ def select_default_proof(
45
+ store: Store, *, accepted_record_ids: Collection[str]
46
+ ) -> ProofSelection | None:
47
+ """Choose one current-run accepted record and its deterministic concrete anchor path."""
48
+ now = datetime.now(UTC)
49
+ eligible_ids = frozenset(accepted_record_ids)
50
+ candidates: list[ProofSelection] = []
51
+ for decision in store.iter_decisions():
52
+ if decision.id not in eligible_ids:
53
+ continue
54
+ if decision.status != DecisionStatus.ACCEPTED:
55
+ continue
56
+ if decision.valid_from > now:
57
+ continue
58
+ if decision.valid_to is not None and decision.valid_to <= now:
59
+ continue
60
+ paths = _live_leaf_paths(store, decision.id)
61
+ if paths:
62
+ candidates.append(
63
+ ProofSelection(decision=decision, file_path=paths[0], rule=_SELECTION_RULE)
64
+ )
65
+ return min(candidates, key=_proof_sort_key) if candidates else None
66
+
67
+
68
+ def prove_task_context(
69
+ store: Store,
70
+ reader: GraphifyReader,
71
+ *,
72
+ accepted_record_ids: Collection[str],
73
+ file_path: str | None = None,
74
+ ) -> ProofResult:
75
+ """Prove that the deterministic record appears through production task retrieval."""
76
+ selection = select_default_proof(store, accepted_record_ids=accepted_record_ids)
77
+ if selection is None:
78
+ return ProofResult(complete=False, reason="no eligible accepted decision")
79
+
80
+ anchor_path = file_path or selection.file_path
81
+ context = get_task_context([Seed(file_path=anchor_path)], store, reader)
82
+ if selection.decision.id not in context.shown_ids:
83
+ return ProofResult(complete=False, reason="selected decision did not surface")
84
+
85
+ rendered = context.render(include_structure=False)
86
+ primary_line = next(
87
+ (line for line in rendered.splitlines() if selection.decision.id in line), None
88
+ )
89
+ if primary_line is None:
90
+ return ProofResult(complete=False, reason="selected decision did not surface")
91
+
92
+ return ProofResult(
93
+ complete=True,
94
+ primary_line=primary_line,
95
+ source=selection.decision.provenance.ref,
96
+ file_path=anchor_path,
97
+ selection_rule=selection.rule,
98
+ full_context=rendered,
99
+ copyable_prompt=(
100
+ f"Call get_task_context for {anchor_path} and explain why record "
101
+ f"{selection.decision.id} applies before editing."
102
+ ),
103
+ )