rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,306 @@
|
|
|
1
|
+
"""CC-5 (compliance §13 C-2): the `compliance_ingestion` subgraph -- regulatory corpus -> Requirement KG.
|
|
2
|
+
|
|
3
|
+
A hardened LangGraph subgraph on `scaffold.py`, following the LG-3 pattern: per § section, extract the deontic
|
|
4
|
+
rules (requirement_extraction, CC-2) and write them as `Requirement` nodes, with retry -> dead-letter per
|
|
5
|
+
section so one bad section never kills the ingest. It REUSES the generic ingestion machinery -- `SourceDocument`,
|
|
6
|
+
the `CorpusAdapter` seam, and the `run_corpus_ingestion` driver (X/N progress + per-doc dead-letter + is_done
|
|
7
|
+
resume) -- via a thin `RegulationAdapter`; only the two per-section stages (extract, write) are compliance-specific.
|
|
8
|
+
|
|
9
|
+
The Requirement KG lives in its OWN database (`ragwright_compliance`), so the contract KG stays clean; the store
|
|
10
|
+
adds the `Requirement` vertex type additively (`ensure_compliance_schema`).
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import asyncio
|
|
16
|
+
|
|
17
|
+
import json
|
|
18
|
+
import re
|
|
19
|
+
from functools import lru_cache
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Any, Callable, Iterable, Optional, TypedDict
|
|
22
|
+
|
|
23
|
+
from langgraph.graph import END, START, StateGraph
|
|
24
|
+
from langgraph.runtime import Runtime
|
|
25
|
+
|
|
26
|
+
from rag_wright.contracts.identifiers import canonical_source_doc_id
|
|
27
|
+
from rag_wright.subgraphs.contract_ingestion_pipeline import (
|
|
28
|
+
IngestionReport,
|
|
29
|
+
SourceDocument,
|
|
30
|
+
arun_corpus_ingestion,
|
|
31
|
+
)
|
|
32
|
+
from rag_wright.subgraphs.scaffold import DEFAULT_RETRY, business_span, dead_letter
|
|
33
|
+
from rag_wright.subgraphs.typed_clause_extraction import TransientExtraction
|
|
34
|
+
|
|
35
|
+
# extract_fn: a section's SourceDocument -> its extracted Requirement[]; write_fn: (doc, reqs) -> count written.
|
|
36
|
+
ExtractReqFn = Callable[[SourceDocument], list]
|
|
37
|
+
WriteReqFn = Callable[[SourceDocument, list], int]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@lru_cache(maxsize=1)
|
|
41
|
+
def _deontic_cue_pattern() -> re.Pattern:
|
|
42
|
+
from rag_wright.ontology.loader import load_deontic_cues
|
|
43
|
+
|
|
44
|
+
cues = sorted(load_deontic_cues(), key=len, reverse=True)
|
|
45
|
+
return re.compile(r"\b(?:" + "|".join(re.escape(c) for c in cues) + r")\b", re.IGNORECASE)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def is_operative(text: str) -> bool:
|
|
49
|
+
"""ADR-0066 P3c (Gap 1): a section states an OPERATIVE rule iff its text carries a deontic CUE (must / shall /
|
|
50
|
+
may / prohibited / ... -- authored in compliance_bridge.ttl `cmp:cue`). A section with NO cue is non-operative
|
|
51
|
+
(a definitions / purpose / scope statement) and is skipped -- the domain-neutral, heading-agnostic replacement
|
|
52
|
+
for the brittle 'definition'-in-heading keyword hack. Recall-first: any cue -> operative -> extracted."""
|
|
53
|
+
return bool(text and _deontic_cue_pattern().search(text))
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class RegulationAdapter:
|
|
57
|
+
"""The per-corpus seam (`CorpusAdapter`) for a regulation: an eCFR-style `sections.json`
|
|
58
|
+
([{section, heading, text}]) -> one `SourceDocument` per § section, carrying the section number and source
|
|
59
|
+
as metadata (the extract stage reads them for the citation). The ONLY regulation-specific code in the path."""
|
|
60
|
+
|
|
61
|
+
def __init__(self, sections_path: Any, source: str, *, limit: int = 0, skip_definitions: bool = True) -> None:
|
|
62
|
+
self._path = sections_path
|
|
63
|
+
self._source = source
|
|
64
|
+
self._limit = limit
|
|
65
|
+
# P3c (Gap 1): skip NON-OPERATIVE sections -- ones with no deontic cue (definitions / purpose / scope).
|
|
66
|
+
# Extracting them over-generates spurious "requirements" (61 from FTC §255.0, ~40% of the KG, a leak
|
|
67
|
+
# surface into judging). `is_operative` is the domain-neutral, ontology-driven cue gate that replaced the
|
|
68
|
+
# brittle "definition"-in-heading keyword hack. `skip_definitions` keeps its name for back-compat.
|
|
69
|
+
self._skip_definitions = skip_definitions
|
|
70
|
+
|
|
71
|
+
def documents(self) -> Iterable[SourceDocument]:
|
|
72
|
+
sections = json.loads(Path(self._path).read_text(encoding="utf-8"))
|
|
73
|
+
if self._limit:
|
|
74
|
+
sections = sections[: self._limit]
|
|
75
|
+
for sec in sections:
|
|
76
|
+
if not sec.get("text", "").strip():
|
|
77
|
+
continue
|
|
78
|
+
if self._skip_definitions and not is_operative(sec.get("text", "")):
|
|
79
|
+
continue # P3c: a section with no deontic cue is non-operative (definitions/purpose) -> skip
|
|
80
|
+
yield SourceDocument(
|
|
81
|
+
source_doc_id=canonical_source_doc_id(f"{self._source}_{sec['section']}"),
|
|
82
|
+
text=sec["text"],
|
|
83
|
+
metadata={"section": sec["section"], "source": self._source},
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class DocumentRegulationAdapter:
|
|
88
|
+
"""DOCPARSE-1 (ADR-0049): the per-corpus seam for a customer's OWN regulation/policy DOCUMENT (PDF/DOCX/HTML),
|
|
89
|
+
not a pre-sectioned eCFR `sections.json`. Parses the document once (docling) and splits it at its headings via
|
|
90
|
+
`document_to_sections`, yielding one `SourceDocument` per section -- the SAME shape `RegulationAdapter` yields,
|
|
91
|
+
so a customer policy PDF flows through the identical compliance pipeline. `sections_fn` is injected (the docling
|
|
92
|
+
parse) so this is hermetically testable; production passes the real `document_to_sections(parse_document_bytes(...))`."""
|
|
93
|
+
|
|
94
|
+
def __init__(self, doc_name: str, data: bytes, source: str, *, sections_fn: Any = None,
|
|
95
|
+
limit: int = 0, skip_definitions: bool = True) -> None:
|
|
96
|
+
self._name = doc_name
|
|
97
|
+
self._data = data
|
|
98
|
+
self._source = source
|
|
99
|
+
self._sections_fn = sections_fn
|
|
100
|
+
self._limit = limit
|
|
101
|
+
self._skip_definitions = skip_definitions
|
|
102
|
+
|
|
103
|
+
def documents(self) -> Iterable[SourceDocument]:
|
|
104
|
+
if self._sections_fn is not None:
|
|
105
|
+
sections = self._sections_fn(self._name, self._data)
|
|
106
|
+
else: # production: docling parse -> heading-split sections (DOCPARSE-1)
|
|
107
|
+
from rag_wright.corpus.document_parser import document_to_sections, parse_document_bytes
|
|
108
|
+
|
|
109
|
+
sections = document_to_sections(parse_document_bytes(self._name, self._data))
|
|
110
|
+
if self._limit:
|
|
111
|
+
sections = sections[: self._limit]
|
|
112
|
+
for i, sec in enumerate(sections, 1):
|
|
113
|
+
if not (sec.get("text") or "").strip():
|
|
114
|
+
continue
|
|
115
|
+
if self._skip_definitions and not is_operative(sec.get("text") or ""):
|
|
116
|
+
continue # P3c: non-operative section (no deontic cue) -> skip
|
|
117
|
+
# a headingless preamble section still ingests -- its citation is its position (never dropped)
|
|
118
|
+
citation = sec.get("section") or str(i)
|
|
119
|
+
yield SourceDocument(
|
|
120
|
+
source_doc_id=canonical_source_doc_id(f"{self._source}_{citation}"),
|
|
121
|
+
text=sec["text"],
|
|
122
|
+
# issue 0043: carry the section's page provenance (from document_to_sections) to the extract stage
|
|
123
|
+
# so each Requirement records its policy page(s). A pre-sectioned corpus (RegulationAdapter) has no
|
|
124
|
+
# parse -> no pages, honestly absent.
|
|
125
|
+
metadata={"section": citation, "source": self._source,
|
|
126
|
+
"pages": sec.get("pages") or [], "bbox": sec.get("bbox")},
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
class ComplianceIngestState(TypedDict, total=False):
|
|
131
|
+
document: SourceDocument
|
|
132
|
+
requirements: list
|
|
133
|
+
written: dict
|
|
134
|
+
dead_letter: Optional[dict]
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def build_compliance_ingest(
|
|
138
|
+
extract_fn: ExtractReqFn, write_fn: WriteReqFn, *, retry_policy: Any = DEFAULT_RETRY
|
|
139
|
+
):
|
|
140
|
+
"""Compile the per-section ingest subgraph: START -> extract[retry] -> write[retry] -> END. Both stages are
|
|
141
|
+
injected for hermetic testing. A stage failure dead-letters the section (dropped with a reason, never
|
|
142
|
+
raised) so one bad section never kills the corpus ingest -- the LG-3 hardening pattern."""
|
|
143
|
+
max_attempts = int(getattr(retry_policy, "max_attempts", 3))
|
|
144
|
+
|
|
145
|
+
async def _aguard(name: str, work: Any, runtime: Runtime, doc: SourceDocument) -> dict:
|
|
146
|
+
attempt = runtime.execution_info.node_attempt
|
|
147
|
+
with business_span(f"compliance_ingestion.{name}", source_doc_id=doc.source_doc_id):
|
|
148
|
+
try:
|
|
149
|
+
return await work()
|
|
150
|
+
except Exception as exc: # noqa: BLE001 - transient -> retry, or dead-letter on exhaustion
|
|
151
|
+
if attempt >= max_attempts:
|
|
152
|
+
return {"dead_letter": dead_letter(
|
|
153
|
+
"ingest_failed", source_doc_id=doc.source_doc_id, stage=name, error=str(exc))}
|
|
154
|
+
raise TransientExtraction(str(exc)) from exc
|
|
155
|
+
|
|
156
|
+
async def extract(state: ComplianceIngestState, runtime: Runtime) -> ComplianceIngestState:
|
|
157
|
+
doc = state["document"]
|
|
158
|
+
|
|
159
|
+
async def _w() -> dict:
|
|
160
|
+
return {"requirements": await extract_fn(doc)}
|
|
161
|
+
|
|
162
|
+
return await _aguard("extract", _w, runtime, doc)
|
|
163
|
+
|
|
164
|
+
async def write(state: ComplianceIngestState, runtime: Runtime) -> ComplianceIngestState:
|
|
165
|
+
if state.get("dead_letter"):
|
|
166
|
+
return {}
|
|
167
|
+
doc = state["document"]
|
|
168
|
+
|
|
169
|
+
async def _w() -> dict:
|
|
170
|
+
return {"written": {"requirements": await write_fn(doc, state.get("requirements", []))}}
|
|
171
|
+
|
|
172
|
+
return await _aguard("write", _w, runtime, doc)
|
|
173
|
+
|
|
174
|
+
g = StateGraph(ComplianceIngestState)
|
|
175
|
+
g.add_node("extract", extract, retry_policy=retry_policy)
|
|
176
|
+
g.add_node("write", write, retry_policy=retry_policy)
|
|
177
|
+
g.add_edge(START, "extract")
|
|
178
|
+
g.add_conditional_edges("extract", lambda s: "end" if s.get("dead_letter") else "write",
|
|
179
|
+
{"write": "write", "end": END})
|
|
180
|
+
g.add_edge("write", END)
|
|
181
|
+
return g.compile()
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def production_compliance_ingestion(store: Any, *, model: Any, extract_override: Optional[ExtractReqFn] = None,
|
|
185
|
+
write_override: Optional[Any] = None, extraction_backend: str = "jev"):
|
|
186
|
+
"""Wire the real capabilities: extract = the requirement_extraction SUBGRAPH (CC-2), write =
|
|
187
|
+
`ComplianceStore(store).write_requirements` (ADR-0117 DD-1b). `extract_override` / `write_override` inject
|
|
188
|
+
stubs for tests.
|
|
189
|
+
|
|
190
|
+
`extraction_backend` (ADR-0119): "jev" (DEFAULT since the corpus A/B -- deterministic `operative_rule_spans` +
|
|
191
|
+
one `jev_decision` call/span for the operative gate + actor + claim_types, cue-deontic, verbatim text; recall
|
|
192
|
+
1.00 vs the rubric gold, ~4.5x cheaper than docling, calibrated, full actor/claim coverage; applicability /
|
|
193
|
+
evidence_standard left empty). REQUIRES the reference pack loaded (`load_reference_pack`, so `jev_decision`
|
|
194
|
+
resolves) + the decision-model key (`OPENROUTER_API_KEY`, or a Laya decisions endpoint via the profile). Set
|
|
195
|
+
"docling" to fall back to the per-section docling-graph LLM extraction (no OpenRouter/Jev dependency)."""
|
|
196
|
+
from rag_wright.capabilities.compliance_store import ComplianceStore
|
|
197
|
+
from rag_wright.capabilities.requirement_extraction import ajev_extract_regulation_section
|
|
198
|
+
from rag_wright.subgraphs.requirement_extraction import run_requirement_extraction
|
|
199
|
+
|
|
200
|
+
req_extract_override = ajev_extract_regulation_section if extraction_backend == "jev" else None
|
|
201
|
+
|
|
202
|
+
async def _extract(doc: SourceDocument) -> list:
|
|
203
|
+
# COMP-ASYNC-1 lossless: raise_on_failure so a FAILED section propagates to the compliance `_aguard`
|
|
204
|
+
# (-> retry -> dead-letter with reason), never silently writing 0 requirements. Genuine-empty still -> [].
|
|
205
|
+
return await run_requirement_extraction(
|
|
206
|
+
doc.text, model=model, source=doc.metadata["source"], section=doc.metadata["section"],
|
|
207
|
+
pages=doc.metadata.get("pages") or [], bbox=doc.metadata.get("bbox"), # issue 0043: policy page(s)
|
|
208
|
+
raise_on_failure=True, extract_override=req_extract_override)
|
|
209
|
+
|
|
210
|
+
write_fn = write_override or ComplianceStore(store).write_requirements
|
|
211
|
+
|
|
212
|
+
async def _awrite(doc: SourceDocument, reqs: list) -> Any:
|
|
213
|
+
return await asyncio.to_thread(write_fn, reqs) # store I/O off the loop
|
|
214
|
+
|
|
215
|
+
return build_compliance_ingest(extract_override or _extract, _awrite)
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def _compliance_is_done(store: Any, source: str) -> Any:
|
|
219
|
+
"""PROD-2 #2 resume: skip a SECTION already ingested for `source` (a present `citation` in the Requirement KG
|
|
220
|
+
-- the compliance analogue of a present `Contract` node). Computed ONCE (one query); a failed/empty section
|
|
221
|
+
wrote no requirement, so it is absent and correctly re-runs. The section's citation is `§ {section}` (matches
|
|
222
|
+
`to_requirements`)."""
|
|
223
|
+
done = store.ingested_citations(source)
|
|
224
|
+
return lambda doc: f"§ {doc.metadata.get('section', '')}" in done
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
async def run_compliance_ingestion(
|
|
228
|
+
sections_path: Any, store: Any, *, model: Any, source: str = "FTC 16 CFR 255",
|
|
229
|
+
extract_override: Optional[ExtractReqFn] = None, write_override: Optional[Any] = None,
|
|
230
|
+
extraction_backend: str = "jev",
|
|
231
|
+
) -> IngestionReport:
|
|
232
|
+
"""Ingest a regulation (`sections.json`) into the Requirement KG: ensure the compliance schema, then map
|
|
233
|
+
every section through the per-section subgraph via the generic corpus driver (X/N progress, per-section
|
|
234
|
+
dead-letter, is_done resume). Point `store` at the compliance database (`ragwright_compliance`).
|
|
235
|
+
`extraction_backend` ("jev" DEFAULT since the corpus A/B | "docling" fallback, ADR-0119) selects the
|
|
236
|
+
requirement-extraction act; "jev" needs `load_reference_pack()` + `OPENROUTER_API_KEY` (see
|
|
237
|
+
`production_compliance_ingestion`)."""
|
|
238
|
+
store.ensure_compliance_schema()
|
|
239
|
+
graph = production_compliance_ingestion(
|
|
240
|
+
store, model=model, extract_override=extract_override, write_override=write_override,
|
|
241
|
+
extraction_backend=extraction_backend)
|
|
242
|
+
return await arun_corpus_ingestion(
|
|
243
|
+
RegulationAdapter(sections_path, source), graph, is_done=_compliance_is_done(store, source))
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
async def run_compliance_document_ingestion(
|
|
247
|
+
doc_name: str, data: bytes, store: Any, *, model: Any, source: str,
|
|
248
|
+
sections_fn: Optional[Any] = None, extract_override: Optional[ExtractReqFn] = None,
|
|
249
|
+
write_override: Optional[Any] = None,
|
|
250
|
+
) -> IngestionReport:
|
|
251
|
+
"""DOCPARSE-1: ingest a customer's OWN regulation/policy DOCUMENT (PDF/DOCX/HTML bytes) into the Requirement
|
|
252
|
+
KG -- the same compliance pipeline, fed by a `DocumentRegulationAdapter` (docling parse -> heading-split
|
|
253
|
+
sections) instead of a pre-sectioned eCFR `sections.json`. `sections_fn` injects the parse for tests."""
|
|
254
|
+
store.ensure_compliance_schema()
|
|
255
|
+
graph = production_compliance_ingestion(
|
|
256
|
+
store, model=model, extract_override=extract_override, write_override=write_override)
|
|
257
|
+
return await arun_corpus_ingestion(
|
|
258
|
+
DocumentRegulationAdapter(doc_name, data, source, sections_fn=sections_fn), graph,
|
|
259
|
+
is_done=_compliance_is_done(store, source))
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def submit_compliance_ingestion(
|
|
263
|
+
adapter: Any, store: Any, jobs: Any, *, job_id: str, model: Any, source: str,
|
|
264
|
+
extract_override: Optional[ExtractReqFn] = None, write_override: Optional[Any] = None,
|
|
265
|
+
max_concurrency: int = 4,
|
|
266
|
+
) -> str:
|
|
267
|
+
"""COMP-ASYNC-1 (ADR-0050): submit an ASYNC compliance ingestion job. Returns `job_id` IMMEDIATELY; sections
|
|
268
|
+
ingest in the background with bounded parallelism, and a FAILED section is dead-lettered on the job (lossless).
|
|
269
|
+
Ensures the compliance schema, builds the per-section graph, then hands the (adapter, graph) to the generic
|
|
270
|
+
async runner. `adapter` = a RegulationAdapter (eCFR sections.json) OR DocumentRegulationAdapter (customer doc).
|
|
271
|
+
`jobs` = the `JobStore`; poll `jobs.get(job_id)` for status."""
|
|
272
|
+
from rag_wright.subgraphs.async_ingestion import submit_ingestion
|
|
273
|
+
|
|
274
|
+
store.ensure_compliance_schema()
|
|
275
|
+
graph = production_compliance_ingestion(
|
|
276
|
+
store, model=model, extract_override=extract_override, write_override=write_override)
|
|
277
|
+
return submit_ingestion(
|
|
278
|
+
adapter, graph, jobs, job_id=job_id, db=getattr(store, "database", ""),
|
|
279
|
+
corpus_ref={"kind": "compliance", "source": source}, max_concurrency=max_concurrency,
|
|
280
|
+
is_done=_compliance_is_done(store, source)) # PROD-2 #2: skip already-ingested sections on resume
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def register_compliance_ingestion(registry) -> None:
|
|
284
|
+
"""Register `compliance_ingestion` (subgraph; CC-5). Contract = `IngestionReport`."""
|
|
285
|
+
registry.register(
|
|
286
|
+
"compliance_ingestion",
|
|
287
|
+
contract=IngestionReport,
|
|
288
|
+
kind="subgraph",
|
|
289
|
+
display_name="Compliance ingestion (regulatory corpus -> Requirement KG)",
|
|
290
|
+
)
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
async def ainvoke(resources, inputs: dict):
|
|
294
|
+
"""EP-REF-1c (ADR-0118): the capability invoke factory (impl_ref target) for policy ingest into the Requirement
|
|
295
|
+
KG. The store + the extraction model (STRUCTURED_REASONING, the quality-sensitive role) come from the workspace
|
|
296
|
+
handle; the policy from `inputs`. Two source shapes: `{source, sections_path}` (a pre-sectioned eCFR-style
|
|
297
|
+
`sections.json`) or `{source, doc_name, data}` (a policy DOCUMENT's raw bytes, split at its headings)."""
|
|
298
|
+
from rag_wright.capabilities.dg_extraction import default_extraction_model
|
|
299
|
+
from rag_wright.models.profiles import ModelRole
|
|
300
|
+
|
|
301
|
+
model = default_extraction_model("requirement-extract", resources.model_id(ModelRole.STRUCTURED_REASONING))
|
|
302
|
+
if inputs.get("data") is not None: # a policy DOCUMENT (bytes)
|
|
303
|
+
return await run_compliance_document_ingestion(
|
|
304
|
+
inputs["doc_name"], inputs["data"], resources._store, model=model, source=inputs["source"])
|
|
305
|
+
return await run_compliance_ingestion( # a pre-sectioned sections.json
|
|
306
|
+
inputs["sections_path"], resources._store, model=model, source=inputs["source"])
|