rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,306 @@
1
+ """CC-5 (compliance §13 C-2): the `compliance_ingestion` subgraph -- regulatory corpus -> Requirement KG.
2
+
3
+ A hardened LangGraph subgraph on `scaffold.py`, following the LG-3 pattern: per § section, extract the deontic
4
+ rules (requirement_extraction, CC-2) and write them as `Requirement` nodes, with retry -> dead-letter per
5
+ section so one bad section never kills the ingest. It REUSES the generic ingestion machinery -- `SourceDocument`,
6
+ the `CorpusAdapter` seam, and the `run_corpus_ingestion` driver (X/N progress + per-doc dead-letter + is_done
7
+ resume) -- via a thin `RegulationAdapter`; only the two per-section stages (extract, write) are compliance-specific.
8
+
9
+ The Requirement KG lives in its OWN database (`ragwright_compliance`), so the contract KG stays clean; the store
10
+ adds the `Requirement` vertex type additively (`ensure_compliance_schema`).
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import asyncio
16
+
17
+ import json
18
+ import re
19
+ from functools import lru_cache
20
+ from pathlib import Path
21
+ from typing import Any, Callable, Iterable, Optional, TypedDict
22
+
23
+ from langgraph.graph import END, START, StateGraph
24
+ from langgraph.runtime import Runtime
25
+
26
+ from rag_wright.contracts.identifiers import canonical_source_doc_id
27
+ from rag_wright.subgraphs.contract_ingestion_pipeline import (
28
+ IngestionReport,
29
+ SourceDocument,
30
+ arun_corpus_ingestion,
31
+ )
32
+ from rag_wright.subgraphs.scaffold import DEFAULT_RETRY, business_span, dead_letter
33
+ from rag_wright.subgraphs.typed_clause_extraction import TransientExtraction
34
+
35
+ # extract_fn: a section's SourceDocument -> its extracted Requirement[]; write_fn: (doc, reqs) -> count written.
36
+ ExtractReqFn = Callable[[SourceDocument], list]
37
+ WriteReqFn = Callable[[SourceDocument, list], int]
38
+
39
+
40
+ @lru_cache(maxsize=1)
41
+ def _deontic_cue_pattern() -> re.Pattern:
42
+ from rag_wright.ontology.loader import load_deontic_cues
43
+
44
+ cues = sorted(load_deontic_cues(), key=len, reverse=True)
45
+ return re.compile(r"\b(?:" + "|".join(re.escape(c) for c in cues) + r")\b", re.IGNORECASE)
46
+
47
+
48
+ def is_operative(text: str) -> bool:
49
+ """ADR-0066 P3c (Gap 1): a section states an OPERATIVE rule iff its text carries a deontic CUE (must / shall /
50
+ may / prohibited / ... -- authored in compliance_bridge.ttl `cmp:cue`). A section with NO cue is non-operative
51
+ (a definitions / purpose / scope statement) and is skipped -- the domain-neutral, heading-agnostic replacement
52
+ for the brittle 'definition'-in-heading keyword hack. Recall-first: any cue -> operative -> extracted."""
53
+ return bool(text and _deontic_cue_pattern().search(text))
54
+
55
+
56
+ class RegulationAdapter:
57
+ """The per-corpus seam (`CorpusAdapter`) for a regulation: an eCFR-style `sections.json`
58
+ ([{section, heading, text}]) -> one `SourceDocument` per § section, carrying the section number and source
59
+ as metadata (the extract stage reads them for the citation). The ONLY regulation-specific code in the path."""
60
+
61
+ def __init__(self, sections_path: Any, source: str, *, limit: int = 0, skip_definitions: bool = True) -> None:
62
+ self._path = sections_path
63
+ self._source = source
64
+ self._limit = limit
65
+ # P3c (Gap 1): skip NON-OPERATIVE sections -- ones with no deontic cue (definitions / purpose / scope).
66
+ # Extracting them over-generates spurious "requirements" (61 from FTC §255.0, ~40% of the KG, a leak
67
+ # surface into judging). `is_operative` is the domain-neutral, ontology-driven cue gate that replaced the
68
+ # brittle "definition"-in-heading keyword hack. `skip_definitions` keeps its name for back-compat.
69
+ self._skip_definitions = skip_definitions
70
+
71
+ def documents(self) -> Iterable[SourceDocument]:
72
+ sections = json.loads(Path(self._path).read_text(encoding="utf-8"))
73
+ if self._limit:
74
+ sections = sections[: self._limit]
75
+ for sec in sections:
76
+ if not sec.get("text", "").strip():
77
+ continue
78
+ if self._skip_definitions and not is_operative(sec.get("text", "")):
79
+ continue # P3c: a section with no deontic cue is non-operative (definitions/purpose) -> skip
80
+ yield SourceDocument(
81
+ source_doc_id=canonical_source_doc_id(f"{self._source}_{sec['section']}"),
82
+ text=sec["text"],
83
+ metadata={"section": sec["section"], "source": self._source},
84
+ )
85
+
86
+
87
+ class DocumentRegulationAdapter:
88
+ """DOCPARSE-1 (ADR-0049): the per-corpus seam for a customer's OWN regulation/policy DOCUMENT (PDF/DOCX/HTML),
89
+ not a pre-sectioned eCFR `sections.json`. Parses the document once (docling) and splits it at its headings via
90
+ `document_to_sections`, yielding one `SourceDocument` per section -- the SAME shape `RegulationAdapter` yields,
91
+ so a customer policy PDF flows through the identical compliance pipeline. `sections_fn` is injected (the docling
92
+ parse) so this is hermetically testable; production passes the real `document_to_sections(parse_document_bytes(...))`."""
93
+
94
+ def __init__(self, doc_name: str, data: bytes, source: str, *, sections_fn: Any = None,
95
+ limit: int = 0, skip_definitions: bool = True) -> None:
96
+ self._name = doc_name
97
+ self._data = data
98
+ self._source = source
99
+ self._sections_fn = sections_fn
100
+ self._limit = limit
101
+ self._skip_definitions = skip_definitions
102
+
103
+ def documents(self) -> Iterable[SourceDocument]:
104
+ if self._sections_fn is not None:
105
+ sections = self._sections_fn(self._name, self._data)
106
+ else: # production: docling parse -> heading-split sections (DOCPARSE-1)
107
+ from rag_wright.corpus.document_parser import document_to_sections, parse_document_bytes
108
+
109
+ sections = document_to_sections(parse_document_bytes(self._name, self._data))
110
+ if self._limit:
111
+ sections = sections[: self._limit]
112
+ for i, sec in enumerate(sections, 1):
113
+ if not (sec.get("text") or "").strip():
114
+ continue
115
+ if self._skip_definitions and not is_operative(sec.get("text") or ""):
116
+ continue # P3c: non-operative section (no deontic cue) -> skip
117
+ # a headingless preamble section still ingests -- its citation is its position (never dropped)
118
+ citation = sec.get("section") or str(i)
119
+ yield SourceDocument(
120
+ source_doc_id=canonical_source_doc_id(f"{self._source}_{citation}"),
121
+ text=sec["text"],
122
+ # issue 0043: carry the section's page provenance (from document_to_sections) to the extract stage
123
+ # so each Requirement records its policy page(s). A pre-sectioned corpus (RegulationAdapter) has no
124
+ # parse -> no pages, honestly absent.
125
+ metadata={"section": citation, "source": self._source,
126
+ "pages": sec.get("pages") or [], "bbox": sec.get("bbox")},
127
+ )
128
+
129
+
130
+ class ComplianceIngestState(TypedDict, total=False):
131
+ document: SourceDocument
132
+ requirements: list
133
+ written: dict
134
+ dead_letter: Optional[dict]
135
+
136
+
137
+ def build_compliance_ingest(
138
+ extract_fn: ExtractReqFn, write_fn: WriteReqFn, *, retry_policy: Any = DEFAULT_RETRY
139
+ ):
140
+ """Compile the per-section ingest subgraph: START -> extract[retry] -> write[retry] -> END. Both stages are
141
+ injected for hermetic testing. A stage failure dead-letters the section (dropped with a reason, never
142
+ raised) so one bad section never kills the corpus ingest -- the LG-3 hardening pattern."""
143
+ max_attempts = int(getattr(retry_policy, "max_attempts", 3))
144
+
145
+ async def _aguard(name: str, work: Any, runtime: Runtime, doc: SourceDocument) -> dict:
146
+ attempt = runtime.execution_info.node_attempt
147
+ with business_span(f"compliance_ingestion.{name}", source_doc_id=doc.source_doc_id):
148
+ try:
149
+ return await work()
150
+ except Exception as exc: # noqa: BLE001 - transient -> retry, or dead-letter on exhaustion
151
+ if attempt >= max_attempts:
152
+ return {"dead_letter": dead_letter(
153
+ "ingest_failed", source_doc_id=doc.source_doc_id, stage=name, error=str(exc))}
154
+ raise TransientExtraction(str(exc)) from exc
155
+
156
+ async def extract(state: ComplianceIngestState, runtime: Runtime) -> ComplianceIngestState:
157
+ doc = state["document"]
158
+
159
+ async def _w() -> dict:
160
+ return {"requirements": await extract_fn(doc)}
161
+
162
+ return await _aguard("extract", _w, runtime, doc)
163
+
164
+ async def write(state: ComplianceIngestState, runtime: Runtime) -> ComplianceIngestState:
165
+ if state.get("dead_letter"):
166
+ return {}
167
+ doc = state["document"]
168
+
169
+ async def _w() -> dict:
170
+ return {"written": {"requirements": await write_fn(doc, state.get("requirements", []))}}
171
+
172
+ return await _aguard("write", _w, runtime, doc)
173
+
174
+ g = StateGraph(ComplianceIngestState)
175
+ g.add_node("extract", extract, retry_policy=retry_policy)
176
+ g.add_node("write", write, retry_policy=retry_policy)
177
+ g.add_edge(START, "extract")
178
+ g.add_conditional_edges("extract", lambda s: "end" if s.get("dead_letter") else "write",
179
+ {"write": "write", "end": END})
180
+ g.add_edge("write", END)
181
+ return g.compile()
182
+
183
+
184
+ def production_compliance_ingestion(store: Any, *, model: Any, extract_override: Optional[ExtractReqFn] = None,
185
+ write_override: Optional[Any] = None, extraction_backend: str = "jev"):
186
+ """Wire the real capabilities: extract = the requirement_extraction SUBGRAPH (CC-2), write =
187
+ `ComplianceStore(store).write_requirements` (ADR-0117 DD-1b). `extract_override` / `write_override` inject
188
+ stubs for tests.
189
+
190
+ `extraction_backend` (ADR-0119): "jev" (DEFAULT since the corpus A/B -- deterministic `operative_rule_spans` +
191
+ one `jev_decision` call/span for the operative gate + actor + claim_types, cue-deontic, verbatim text; recall
192
+ 1.00 vs the rubric gold, ~4.5x cheaper than docling, calibrated, full actor/claim coverage; applicability /
193
+ evidence_standard left empty). REQUIRES the reference pack loaded (`load_reference_pack`, so `jev_decision`
194
+ resolves) + the decision-model key (`OPENROUTER_API_KEY`, or a Laya decisions endpoint via the profile). Set
195
+ "docling" to fall back to the per-section docling-graph LLM extraction (no OpenRouter/Jev dependency)."""
196
+ from rag_wright.capabilities.compliance_store import ComplianceStore
197
+ from rag_wright.capabilities.requirement_extraction import ajev_extract_regulation_section
198
+ from rag_wright.subgraphs.requirement_extraction import run_requirement_extraction
199
+
200
+ req_extract_override = ajev_extract_regulation_section if extraction_backend == "jev" else None
201
+
202
+ async def _extract(doc: SourceDocument) -> list:
203
+ # COMP-ASYNC-1 lossless: raise_on_failure so a FAILED section propagates to the compliance `_aguard`
204
+ # (-> retry -> dead-letter with reason), never silently writing 0 requirements. Genuine-empty still -> [].
205
+ return await run_requirement_extraction(
206
+ doc.text, model=model, source=doc.metadata["source"], section=doc.metadata["section"],
207
+ pages=doc.metadata.get("pages") or [], bbox=doc.metadata.get("bbox"), # issue 0043: policy page(s)
208
+ raise_on_failure=True, extract_override=req_extract_override)
209
+
210
+ write_fn = write_override or ComplianceStore(store).write_requirements
211
+
212
+ async def _awrite(doc: SourceDocument, reqs: list) -> Any:
213
+ return await asyncio.to_thread(write_fn, reqs) # store I/O off the loop
214
+
215
+ return build_compliance_ingest(extract_override or _extract, _awrite)
216
+
217
+
218
+ def _compliance_is_done(store: Any, source: str) -> Any:
219
+ """PROD-2 #2 resume: skip a SECTION already ingested for `source` (a present `citation` in the Requirement KG
220
+ -- the compliance analogue of a present `Contract` node). Computed ONCE (one query); a failed/empty section
221
+ wrote no requirement, so it is absent and correctly re-runs. The section's citation is `§ {section}` (matches
222
+ `to_requirements`)."""
223
+ done = store.ingested_citations(source)
224
+ return lambda doc: f"§ {doc.metadata.get('section', '')}" in done
225
+
226
+
227
+ async def run_compliance_ingestion(
228
+ sections_path: Any, store: Any, *, model: Any, source: str = "FTC 16 CFR 255",
229
+ extract_override: Optional[ExtractReqFn] = None, write_override: Optional[Any] = None,
230
+ extraction_backend: str = "jev",
231
+ ) -> IngestionReport:
232
+ """Ingest a regulation (`sections.json`) into the Requirement KG: ensure the compliance schema, then map
233
+ every section through the per-section subgraph via the generic corpus driver (X/N progress, per-section
234
+ dead-letter, is_done resume). Point `store` at the compliance database (`ragwright_compliance`).
235
+ `extraction_backend` ("jev" DEFAULT since the corpus A/B | "docling" fallback, ADR-0119) selects the
236
+ requirement-extraction act; "jev" needs `load_reference_pack()` + `OPENROUTER_API_KEY` (see
237
+ `production_compliance_ingestion`)."""
238
+ store.ensure_compliance_schema()
239
+ graph = production_compliance_ingestion(
240
+ store, model=model, extract_override=extract_override, write_override=write_override,
241
+ extraction_backend=extraction_backend)
242
+ return await arun_corpus_ingestion(
243
+ RegulationAdapter(sections_path, source), graph, is_done=_compliance_is_done(store, source))
244
+
245
+
246
+ async def run_compliance_document_ingestion(
247
+ doc_name: str, data: bytes, store: Any, *, model: Any, source: str,
248
+ sections_fn: Optional[Any] = None, extract_override: Optional[ExtractReqFn] = None,
249
+ write_override: Optional[Any] = None,
250
+ ) -> IngestionReport:
251
+ """DOCPARSE-1: ingest a customer's OWN regulation/policy DOCUMENT (PDF/DOCX/HTML bytes) into the Requirement
252
+ KG -- the same compliance pipeline, fed by a `DocumentRegulationAdapter` (docling parse -> heading-split
253
+ sections) instead of a pre-sectioned eCFR `sections.json`. `sections_fn` injects the parse for tests."""
254
+ store.ensure_compliance_schema()
255
+ graph = production_compliance_ingestion(
256
+ store, model=model, extract_override=extract_override, write_override=write_override)
257
+ return await arun_corpus_ingestion(
258
+ DocumentRegulationAdapter(doc_name, data, source, sections_fn=sections_fn), graph,
259
+ is_done=_compliance_is_done(store, source))
260
+
261
+
262
+ def submit_compliance_ingestion(
263
+ adapter: Any, store: Any, jobs: Any, *, job_id: str, model: Any, source: str,
264
+ extract_override: Optional[ExtractReqFn] = None, write_override: Optional[Any] = None,
265
+ max_concurrency: int = 4,
266
+ ) -> str:
267
+ """COMP-ASYNC-1 (ADR-0050): submit an ASYNC compliance ingestion job. Returns `job_id` IMMEDIATELY; sections
268
+ ingest in the background with bounded parallelism, and a FAILED section is dead-lettered on the job (lossless).
269
+ Ensures the compliance schema, builds the per-section graph, then hands the (adapter, graph) to the generic
270
+ async runner. `adapter` = a RegulationAdapter (eCFR sections.json) OR DocumentRegulationAdapter (customer doc).
271
+ `jobs` = the `JobStore`; poll `jobs.get(job_id)` for status."""
272
+ from rag_wright.subgraphs.async_ingestion import submit_ingestion
273
+
274
+ store.ensure_compliance_schema()
275
+ graph = production_compliance_ingestion(
276
+ store, model=model, extract_override=extract_override, write_override=write_override)
277
+ return submit_ingestion(
278
+ adapter, graph, jobs, job_id=job_id, db=getattr(store, "database", ""),
279
+ corpus_ref={"kind": "compliance", "source": source}, max_concurrency=max_concurrency,
280
+ is_done=_compliance_is_done(store, source)) # PROD-2 #2: skip already-ingested sections on resume
281
+
282
+
283
+ def register_compliance_ingestion(registry) -> None:
284
+ """Register `compliance_ingestion` (subgraph; CC-5). Contract = `IngestionReport`."""
285
+ registry.register(
286
+ "compliance_ingestion",
287
+ contract=IngestionReport,
288
+ kind="subgraph",
289
+ display_name="Compliance ingestion (regulatory corpus -> Requirement KG)",
290
+ )
291
+
292
+
293
+ async def ainvoke(resources, inputs: dict):
294
+ """EP-REF-1c (ADR-0118): the capability invoke factory (impl_ref target) for policy ingest into the Requirement
295
+ KG. The store + the extraction model (STRUCTURED_REASONING, the quality-sensitive role) come from the workspace
296
+ handle; the policy from `inputs`. Two source shapes: `{source, sections_path}` (a pre-sectioned eCFR-style
297
+ `sections.json`) or `{source, doc_name, data}` (a policy DOCUMENT's raw bytes, split at its headings)."""
298
+ from rag_wright.capabilities.dg_extraction import default_extraction_model
299
+ from rag_wright.models.profiles import ModelRole
300
+
301
+ model = default_extraction_model("requirement-extract", resources.model_id(ModelRole.STRUCTURED_REASONING))
302
+ if inputs.get("data") is not None: # a policy DOCUMENT (bytes)
303
+ return await run_compliance_document_ingestion(
304
+ inputs["doc_name"], inputs["data"], resources._store, model=model, source=inputs["source"])
305
+ return await run_compliance_ingestion( # a pre-sectioned sections.json
306
+ inputs["sections_path"], resources._store, model=model, source=inputs["source"])