rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,165 @@
1
+ """LG-3a: `relational_qa` as a composite LangGraph subgraph -- the REFERENCE composite.
2
+
3
+ Answers an entity question with a grounded, cited answer built from the GRAPH STRUCTURE itself (the standing
4
+ rule: the graph is the relationship layer). No chunk-text rehydration: a relational fact IS the traversed edge.
5
+
6
+ START --> traverse [RetryPolicy] (graph_query: entity -> cited relationship evidence, FR-C.5)
7
+ |
8
+ v
9
+ assemble (graph_structural_evidence: each reached entity -> a cited relationship statement)
10
+ |
11
+ v
12
+ generate (generate_answer: grounded/cited/abstaining answer, FR-Q.6) --> END
13
+
14
+ Why graph-structural, not chunk text: the CONTRACTS_WITH edges are co-party facts extracted from the contract
15
+ preamble; their provenance `chunk_id` is a whole-document id whose text is not stored for retrieval. Rehydrating
16
+ it would need a throwaway text store, and citing "a" same-relationship span would be a one-to-many guess. Instead
17
+ the evidence is the relationship + the target entity, cited by the SOURCE CONTRACT (parsed from the edge's
18
+ provenance chunk_id) -- verifiable and faithful, with nothing to rehydrate (A2, MCP-PROTO Phase A).
19
+
20
+ Query-side posture (matches `query_constraint_extraction`, LG-2a): hardening applied JUDICIOUSLY.
21
+ - **traverse** is the only retried node (a graph/store IO blip is transient). On retry exhaustion it DEGRADES
22
+ to empty evidence rather than dead-lettering -- the generator then abstains, so the query survives.
23
+ - **assemble** is a pure, deterministic graph->evidence step (no store read -> no orphan, no dead_letter).
24
+ - **generate** enforces no-claim-without-a-citation in code (FR-Q.6): empty evidence abstains with no model
25
+ call; fabricated citations are dropped. Confidence tags surfaced by graph_query (FR-S.4) thread through.
26
+
27
+ `traverse_fn` / `generate_fn` are dependency-injected so the graph is hermetically testable with stubs.
28
+ `production_relational_qa` wires the real `graph_query` + `generate_answer`.
29
+ """
30
+
31
+ from __future__ import annotations
32
+
33
+ from typing import Any, Awaitable, Callable, Optional, TypedDict
34
+
35
+ from langgraph.graph import END, START, StateGraph
36
+ from langgraph.runtime import Runtime
37
+
38
+ from rag_wright.capabilities.answer_generator import EvidenceItem, GeneratedAnswer
39
+ from rag_wright.capabilities.graph_query import GraphAnswer
40
+ from rag_wright.ontology.contract_taxonomy import CONTRACTS_WITH # this is a CONTRACT-reference leg (DD-5)
41
+ from rag_wright.subgraphs.scaffold import DEFAULT_RETRY, business_span
42
+ from rag_wright.subgraphs.typed_clause_extraction import TransientExtraction # shared retryable-blip signal
43
+
44
+ # traverse_fn: (start_entity_id, relationship_type, max_hops) -> GraphAnswer (must raise on a transient blip).
45
+ # DD-5: relationship_type is an opaque edge-type string the caller names (graph_query is domain-free).
46
+ TraverseFn = Callable[[str, str, int], GraphAnswer]
47
+ # ASYNC-C1 (ADR-0057): generate_fn is async (the model call gets a true wall-clock deadline via the seam).
48
+ GenerateFn = Callable[[str, list[EvidenceItem]], Awaitable[GeneratedAnswer]]
49
+
50
+ DEFAULT_RELATIONSHIP = CONTRACTS_WITH # the contract co-party edge (this is the contract-reference leg)
51
+
52
+
53
+ class RelationalQAState(TypedDict, total=False):
54
+ query: str
55
+ start_entity_id: str
56
+ relationship_type: str
57
+ max_hops: int
58
+ graph_answer: GraphAnswer
59
+ evidence: list[EvidenceItem]
60
+ answer: GeneratedAnswer
61
+ dead_letter: Optional[dict]
62
+
63
+
64
+ def graph_structural_evidence(graph_answer: GraphAnswer) -> list[EvidenceItem]:
65
+ """The relational evidence IS the graph structure. Each reached entity yields one cited fact per source
66
+ contract: "<start> <rel> <target> (per <contract>)", cited by the CONTRACT id parsed from the edge's
67
+ provenance chunk_id (`<source_doc_id>:<index>:<hash>` -> source_doc_id = contract_id) -- verifiable, with no
68
+ chunk text to rehydrate. Confidence = the path's first surfaced edge tag. Falls back to citing the target
69
+ entity_id when an edge carries no chunk provenance. First-wins/dedup on (target, citation)."""
70
+ start = graph_answer.start_entity_id
71
+ rel = graph_answer.relationship_type
72
+ items: list[EvidenceItem] = []
73
+ seen: set[tuple[str, str]] = set()
74
+ for ev in graph_answer.evidence:
75
+ conf = ev.confidences[0] if ev.confidences else None
76
+ contracts = sorted({cid.split(":")[0] for cid in ev.chunk_ids if cid}) or [ev.entity_id]
77
+ for cite in contracts:
78
+ key = (ev.entity_id, cite)
79
+ if key in seen:
80
+ continue
81
+ seen.add(key)
82
+ items.append(EvidenceItem(
83
+ chunk_id=cite,
84
+ text=f"{start} {rel} {ev.name} (entity {ev.entity_id}; per {cite})",
85
+ confidence=conf))
86
+ return items
87
+
88
+
89
+ def build_relational_qa(traverse_fn: TraverseFn, generate_fn: GenerateFn, *, retry_policy: Any = DEFAULT_RETRY):
90
+ """Compile the `relational_qa` subgraph. `traverse_fn` / `generate_fn` are injected for hermetic testing;
91
+ the graph->evidence assembly is a pure deterministic node. `retry_policy` is the traverse node's policy."""
92
+ max_attempts = int(getattr(retry_policy, "max_attempts", 3))
93
+
94
+ def traverse(state: RelationalQAState, runtime: Runtime) -> RelationalQAState:
95
+ # node_attempt is 1-indexed; a transient blip re-raises so the RetryPolicy retries, EXCEPT on the
96
+ # final attempt where it degrades to EMPTY evidence (the generator abstains -- the query is never lost).
97
+ attempt = runtime.execution_info.node_attempt
98
+ start = state["start_entity_id"]
99
+ rel = state.get("relationship_type", DEFAULT_RELATIONSHIP)
100
+ with business_span("relational_qa.traverse", start_entity_id=start):
101
+ try:
102
+ graph_answer = traverse_fn(start, rel, state.get("max_hops", 1))
103
+ except Exception as exc: # noqa: BLE001 - transient -> retry, or degrade to empty on exhaustion
104
+ if attempt >= max_attempts:
105
+ return {"graph_answer": GraphAnswer(
106
+ start_entity_id=start, relationship_type=rel, evidence=[])}
107
+ raise TransientExtraction(str(exc)) from exc
108
+ return {"graph_answer": graph_answer}
109
+
110
+ def assemble(state: RelationalQAState) -> RelationalQAState:
111
+ with business_span("relational_qa.assemble"):
112
+ return {"evidence": graph_structural_evidence(state["graph_answer"])}
113
+
114
+ async def generate(state: RelationalQAState) -> RelationalQAState:
115
+ with business_span("relational_qa.generate"):
116
+ return {"answer": await generate_fn(state["query"], state.get("evidence", []))}
117
+
118
+ g = StateGraph(RelationalQAState)
119
+ g.add_node("traverse", traverse, retry_policy=retry_policy)
120
+ g.add_node("assemble", assemble)
121
+ g.add_node("generate", generate)
122
+ g.add_edge(START, "traverse")
123
+ g.add_edge("traverse", "assemble")
124
+ g.add_edge("assemble", "generate")
125
+ g.add_edge("generate", END)
126
+ return g.compile()
127
+
128
+
129
+ def production_relational_qa(*, store: Any, answer_model: Any):
130
+ """Wire the real `graph_query` + `generate_answer` into the composite (no text_store: the evidence is
131
+ graph-structural). Imports are lazy so the module stays import-light and hermetic (tests inject stubs)."""
132
+ from rag_wright.capabilities.answer_generator import agenerate_answer
133
+ from rag_wright.capabilities.graph_query import graph_query
134
+
135
+ def traverse(start: str, rel: str, max_hops: int) -> GraphAnswer:
136
+ # graph_query is domain-free and takes a generic edge-type string; this contract-reference leg passes the
137
+ # contract edge value (the contract vocab stays on the caller side, not in the generic primitive).
138
+ return graph_query(start, store=store, relationship_type=rel, max_hops=max_hops)
139
+
140
+ async def generate(query: str, evidence: list[EvidenceItem]) -> GeneratedAnswer:
141
+ return await agenerate_answer(query, evidence, model=answer_model)
142
+
143
+ return build_relational_qa(traverse, generate)
144
+
145
+
146
+ def register_relational_qa(registry) -> None:
147
+ """LG-3a: register `relational_qa` (composite subgraph; graph_query -> graph-structural evidence ->
148
+ generate_answer)."""
149
+ registry.register(
150
+ "relational_qa",
151
+ contract=GeneratedAnswer,
152
+ kind="subgraph",
153
+ display_name="Relational QA (cited answer from graph traversal)",
154
+ )
155
+
156
+
157
+ async def ainvoke(resources, inputs: dict):
158
+ """EP-CORE-2 (ADR-0118): the capability invoke factory (impl_ref target)."""
159
+ from rag_wright.capabilities.answer_generator import answer_model_for
160
+ from rag_wright.models.profiles import ModelRole
161
+
162
+ graph = production_relational_qa(store=resources._store,
163
+ answer_model=answer_model_for(resources.model_id(ModelRole.GENERAL)))
164
+ return await graph.ainvoke({"query": inputs["query"], "start_entity_id": inputs["start_entity_id"],
165
+ "max_hops": inputs.get("max_hops", 1)})
@@ -0,0 +1,137 @@
1
+ """CC-2 (compliance §13), SKILL-SPLIT: the `requirement_extraction` SUBGRAPH.
2
+
3
+ `requirement_extraction` is a SUBGRAPH, not a single-shot skill, because its docling-graph `auto/dense`
4
+ extraction is MULTI-LLM-call (skeleton-then-fill) and the extract -> adapt chaining is a deterministic workflow
5
+ (the rubric's rationale for a subgraph). A hardened LangGraph on `scaffold.py`:
6
+
7
+ START --> extract [RetryPolicy] (the extraction ACT: docling-graph fills the skill's template.py schema)
8
+ | transient failure -> retry, exhaustion -> dead_letter
9
+ v
10
+ adapt --> END (the requirement_adaptation FUNCTION: raw section -> validated Requirement[])
11
+
12
+ `extract`/`adapt` are DI'd for hermetic tests. `run_requirement_extraction` invokes the compiled graph for a
13
+ single section (used by the `compliance_ingestion` corpus driver, CC-5).
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ from typing import Any, Callable, Optional, TypedDict
19
+
20
+ from langgraph.graph import END, START, StateGraph
21
+ from langgraph.runtime import Runtime
22
+
23
+ from rag_wright.capabilities.requirement_extraction import aextract_regulation_section, to_requirements
24
+ from rag_wright.contracts.compliance import Requirement
25
+ from rag_wright.subgraphs.scaffold import DEFAULT_RETRY, business_span, dead_letter
26
+ from rag_wright.subgraphs.typed_clause_extraction import TransientExtraction
27
+
28
+ # extract_fn: section text -> raw ExtractedRegulationSection | None; adapt_fn: (extracted, source, section) -> Requirement[]
29
+ ExtractSectionFn = Callable[[str], Any]
30
+ AdaptFn = Callable[[Any, str, str], list]
31
+
32
+
33
+ class RequirementExtractionFailed(Exception):
34
+ """COMP-ASYNC-1 (PROD-3 lossless, ADR-0050): the per-section requirement extraction FAILED (its subgraph
35
+ dead-lettered after retries) -- distinct from a genuine section with no requirements. Raised so the compliance
36
+ corpus driver can dead-letter the SECTION (visible in the report/job) instead of silently writing 0
37
+ requirements. `section`/`reason` carry the citation + the underlying error."""
38
+
39
+ def __init__(self, section: str, reason: str) -> None:
40
+ super().__init__(f"requirement extraction failed for section {section}: {reason}")
41
+ self.section = section
42
+ self.reason = reason
43
+
44
+
45
+ class ReqExtractState(TypedDict, total=False):
46
+ text: str
47
+ source: str
48
+ section: str
49
+ pages: list # issue 0043: the section's source page(s), stamped onto each extracted Requirement
50
+ bbox: Any # issue 0043: best-effort (l, t, r, b) for a single-item section, else None
51
+ extracted: Any
52
+ requirements: list
53
+ dead_letter: Optional[dict]
54
+
55
+
56
+ def build_requirement_extraction(
57
+ extract_fn: ExtractSectionFn, adapt_fn: AdaptFn, *, retry_policy: Any = DEFAULT_RETRY
58
+ ):
59
+ """Compile the requirement-extraction subgraph: extract[retry] -> adapt -> END. Both nodes DI'd for tests.
60
+ A transient extraction failure retries, then dead-letters the section (never raised) -> an empty result."""
61
+ max_attempts = int(getattr(retry_policy, "max_attempts", 3))
62
+
63
+ async def extract(state: ReqExtractState, runtime: Runtime) -> ReqExtractState:
64
+ section = state.get("section", "")
65
+ attempt = runtime.execution_info.node_attempt
66
+ with business_span("requirement_extraction.extract", section=section):
67
+ try:
68
+ return {"extracted": await extract_fn(state["text"])} # ASYNC (ADR-0057): async extract seam
69
+ except Exception as exc: # noqa: BLE001 - transient -> retry, or dead-letter on exhaustion
70
+ if attempt >= max_attempts:
71
+ return {"dead_letter": dead_letter(
72
+ "extract_failed", section=section, stage="extract", error=str(exc))}
73
+ raise TransientExtraction(str(exc)) from exc
74
+
75
+ async def adapt(state: ReqExtractState) -> ReqExtractState:
76
+ extracted = state.get("extracted")
77
+ if state.get("dead_letter") or extracted is None:
78
+ return {"requirements": []}
79
+ with business_span("requirement_extraction.adapt"):
80
+ reqs = adapt_fn(extracted, state["source"], state["section"]) # adapt is sync/CPU
81
+ # issue 0043: stamp the section's page provenance onto each Requirement (pages/bbox are not part of the
82
+ # requirement_id, so this never affects identity/idempotency). Empty pages when the corpus has no parse.
83
+ pages, bbox = list(state.get("pages") or []), state.get("bbox")
84
+ if pages or bbox is not None:
85
+ reqs = [r.model_copy(update={"pages": pages, "bbox": bbox}) for r in reqs]
86
+ return {"requirements": reqs}
87
+
88
+ g = StateGraph(ReqExtractState)
89
+ g.add_node("extract", extract, retry_policy=retry_policy)
90
+ g.add_node("adapt", adapt)
91
+ g.add_edge(START, "extract")
92
+ g.add_edge("extract", "adapt") # adapt handles the dead-letter/None case (-> [])
93
+ g.add_edge("adapt", END)
94
+ return g.compile()
95
+
96
+
97
+ def production_requirement_extraction(
98
+ *, model: Any, extraction_contract: str = "auto", extract_override: Optional[ExtractSectionFn] = None
99
+ ):
100
+ """Wire the real capabilities: extract = the docling-graph extraction ACT (skills/requirement_extraction/),
101
+ adapt = the `to_requirements` FUNCTION. `extract_override` injects a stub for hermetic driver tests."""
102
+ async def _extract(text: str) -> Any:
103
+ return await aextract_regulation_section(text, model=model, extraction_contract=extraction_contract)
104
+
105
+ return build_requirement_extraction(
106
+ extract_override or _extract,
107
+ lambda extracted, source, section: to_requirements(extracted, source=source, section=section),
108
+ )
109
+
110
+
111
+ async def run_requirement_extraction(
112
+ text: str, *, model: Any, source: str, section: str, extract_override: Optional[ExtractSectionFn] = None,
113
+ pages: Optional[list] = None, bbox: Any = None, raise_on_failure: bool = False,
114
+ ) -> list[Requirement]:
115
+ """Invoke the requirement-extraction subgraph for one § section -> its `Requirement[]`.
116
+
117
+ Default (`raise_on_failure=False`, back-compat): `[]` whether the section genuinely has no requirements OR the
118
+ extraction dead-lettered. COMP-ASYNC-1 lossless: with `raise_on_failure=True`, a dead-letter (a real FAILURE,
119
+ not a genuine-empty section) RAISES `RequirementExtractionFailed` so the compliance corpus driver dead-letters
120
+ the section instead of silently writing 0 requirements. A genuine-empty section still returns []."""
121
+ graph = production_requirement_extraction(model=model, extract_override=extract_override)
122
+ result = await graph.ainvoke({"text": text, "source": source, "section": section,
123
+ "pages": pages or [], "bbox": bbox}) # issue 0043: policy page provenance
124
+ if raise_on_failure and result.get("dead_letter"):
125
+ raise RequirementExtractionFailed(section, str(result["dead_letter"].get("error", "extraction failed")))
126
+ return result.get("requirements", [])
127
+
128
+
129
+ def register_requirement_extraction(registry) -> None:
130
+ """Register `requirement_extraction` as a SUBGRAPH (CC-2): the extract -> adapt workflow over one section.
131
+ Contract = `Requirement`."""
132
+ registry.register(
133
+ "requirement_extraction",
134
+ contract=Requirement,
135
+ kind="subgraph",
136
+ display_name="Requirement extraction (regulatory section -> deontic rules; subgraph)",
137
+ )
@@ -0,0 +1,65 @@
1
+ """LG-0: the reusable pattern + primitives for hardened subgraph capabilities (LangGraph).
2
+
3
+ A `subgraph` capability is a compiled `StateGraph` invoked as a subagent under a contract. Every RAG_Wright
4
+ subgraph is built the same way so hardening is uniform and hermetic-testable:
5
+
6
+ - **Dependency-injected model/store** (no global clients) -> tests inject fakes, no live LLM/DB.
7
+ - A **RetryPolicy** (`DEFAULT_RETRY`) on LLM / IO nodes for transient failures (network, rate limit, 5xx).
8
+ Configure `retry_on` per node for that node's real transient exceptions -- LangGraph's default retry_on
9
+ does NOT retry `ValueError`/`OSError`/etc., so a provider error mapped to one of those needs an explicit
10
+ `retry_on` (or a dedicated transient exception).
11
+ - A conditional **ESCALATION** edge (cheap model -> stronger model, ADR-0028 Flash->Pro), bounded by an
12
+ attempt counter in state -- this is a routing decision, distinct from a retry of the same node.
13
+ - An optional `interrupt()`-based **HUMAN GATE** for low-confidence / high-stakes items.
14
+ - A **DEAD-LETTER** terminal: a bad item is dropped with a reason (never raised), so one item never kills a
15
+ batch (the "17% dead-lettered" lesson).
16
+
17
+ **Observability (GraphWright contract, `temp/observability-contract.md`).** We do NOT build tracing, token
18
+ counting, or a Langfuse client -- GraphWright installs global OpenTelemetry instrumentation (exporting to
19
+ Langfuse or any swappable OTLP backend). Requirements every subgraph MUST meet to be observable for free:
20
+ - construct EVERY model through the LangChain seam (`models/seam.py::build_model`/`build_structured` ->
21
+ `ChatOpenAI`); those calls are auto-captured with token counts + latency;
22
+ - a **raw-SDK** call that bypasses LangChain (notably **docling-graph / LiteLLM** in `dg_extraction`, or any
23
+ sandboxed model call) is INVISIBLE to tracing -- wrap it in `observability.raw_llm_span(...)` +
24
+ `record_tokens(...)` on the ambient tracer (never a new provider);
25
+ - use `observability.business_span(...)` for optional domain spans; never manually span a LangChain call;
26
+ - propagate context across a boundary you introduce (separate process/worker/queue/custom async loop) with
27
+ `observability.inject_context` / `attach_context` -- std-lib threads are already carried by GraphWright.
28
+ See `rag_wright.subgraphs.observability` (a dependency-free no-op seam when OTel is absent, e.g. in tests).
29
+
30
+ `typed_clause_extraction` (LG-1) is the reference implementation; other subgraphs follow this shape.
31
+ Grounding: the LangChain docs MCP (`mcp__docs-langchain__*`) + the framework index (langgraph AST); OTel via
32
+ GraphWright's runtime.
33
+ """
34
+
35
+ from __future__ import annotations
36
+
37
+ from typing import Any
38
+
39
+ from langgraph.types import RetryPolicy
40
+
41
+ # Re-exported so a subgraph author imports the pattern from one place. These wrap the AMBIENT OTel tracer per
42
+ # the GraphWright observability contract; they never construct a provider/exporter or a Langfuse client.
43
+ from rag_wright.subgraphs.observability import ( # noqa: F401
44
+ attach_context,
45
+ business_span,
46
+ inject_context,
47
+ otel_active,
48
+ raw_llm_span,
49
+ record_tokens,
50
+ )
51
+
52
+ # The shared retry for LLM / IO nodes. Backoff on transient failures; bounded attempts. Note LangGraph's
53
+ # default `retry_on` already skips programmer errors (TypeError/ValueError/ImportError/...) and only retries
54
+ # 5xx for requests/httpx -- so a node whose transient error surfaces as one of those must pass its own
55
+ # `retry_on` when it adds the node (see the reference subgraph / LG-1).
56
+ DEFAULT_RETRY = RetryPolicy(max_attempts=3, initial_interval=1.0)
57
+
58
+
59
+ def dead_letter(reason: str, **fields: Any) -> dict:
60
+ """A terminal dead-letter record: the item is dropped with a `reason` (not raised) so the batch survives.
61
+
62
+ Put it on the subgraph state's `dead_letter` key; the caller filters out items that carry one. Extra
63
+ `fields` capture context (the offending id, the exception text, the node) for diagnosis.
64
+ """
65
+ return {"reason": reason, **fields}
@@ -0,0 +1,183 @@
1
+ """LG-2: `semantic_chunking` as a hardened, GRANULAR LangGraph subgraph.
2
+
3
+ The single-call (non-agentic) chunker, expressed as explicit nodes -- the multi-step pipeline is the graph,
4
+ not a hidden orchestrator:
5
+
6
+ gate --cached--> END
7
+ | (miss: load_document)
8
+ v
9
+ discover [RetryPolicy + error_handler -> dead-letter] --dead_letter--> END
10
+ v
11
+ finalize (validate partition + _finalize_chunks) --dead_letter--> END
12
+ v
13
+ summarize (async, concurrent per-chunk summaries)
14
+ v
15
+ manifest (build chunks + validate boundaries + write cache) --> END
16
+
17
+ - **gate**: content-hash short-circuit -- a cached manifest is returned with no LLM call.
18
+ - **discover**: the ONE LLM step (single-call boundary discovery). A transient blip is retried
19
+ (`TransientExtraction` -> DEFAULT_RETRY); on retry exhaustion the `error_handler` dead-letters the document
20
+ (skip it -- never lose the batch) rather than raising.
21
+ - **finalize / manifest**: the deterministic layer, reused verbatim from `rlm_chunking` (`_finalize_chunks`,
22
+ `_validate_partition`, `_validate_boundaries`, `ChunkManifest`).
23
+ - **summarize**: a sync node using `asyncio.run` for the concurrent per-chunk summaries (matches
24
+ `rlm_chunking.chunk`). `asyncio.run` copies the current contextvars context, so the OTel trace context is
25
+ preserved (it is NOT a context-detached loop); the summarizer goes through the LangChain seam
26
+ (auto-captured), and its `to_thread` fan-out is carried by GraphWright's threading instrumentation.
27
+ - **observability**: `discover` is wrapped in `raw_llm_span` (raw-SDK), `summarize` in `business_span`.
28
+
29
+ `discoverer` / `summarizer` are the same injected seams as `rlm_chunking.chunk`, so the graph is hermetically
30
+ testable with the existing stubs. The default discoverer is the deterministic single-call one.
31
+ """
32
+
33
+ from __future__ import annotations
34
+
35
+ import asyncio
36
+ from pathlib import Path
37
+ from typing import Any, Optional, TypedDict
38
+
39
+ from langgraph.graph import END, START, StateGraph
40
+ from langgraph.runtime import Runtime
41
+
42
+ from rag_wright.capabilities.parsing import ParsedDocument, load_document
43
+ from rag_wright.capabilities.rlm_chunking import (
44
+ DEFAULT_SUMMARY_CONCURRENCY,
45
+ DEFAULT_TOKEN_CAP,
46
+ BoundaryDiscoverer,
47
+ BoundaryValidationError,
48
+ Chunk,
49
+ ChunkManifest,
50
+ SeamSummarizer,
51
+ SingleCallBoundaryDiscoverer,
52
+ Summarizer,
53
+ _chunk_offsets,
54
+ _estimate_tokens,
55
+ _finalize_chunks,
56
+ _summarize_all,
57
+ _validate_boundaries,
58
+ _validate_partition,
59
+ )
60
+ from rag_wright.contracts.identifiers import ChunkId
61
+ from rag_wright.subgraphs.scaffold import DEFAULT_RETRY, business_span, dead_letter, raw_llm_span
62
+ from rag_wright.subgraphs.typed_clause_extraction import TransientExtraction
63
+
64
+
65
+ class SemanticChunkingState(TypedDict, total=False):
66
+ parsed: ParsedDocument
67
+ cache_dir: Path
68
+ token_cap: int
69
+ max_concurrency: int
70
+ document: Any
71
+ spans: list
72
+ texts: list
73
+ offsets: list
74
+ summaries: list
75
+ manifest: Optional[ChunkManifest]
76
+ dead_letter: Optional[dict]
77
+
78
+
79
+ def _manifest_path(state: SemanticChunkingState) -> Path:
80
+ parsed = state["parsed"]
81
+ return state["cache_dir"] / f"{parsed.source_doc_id}.{parsed.content_hash[:16]}.chunks.json"
82
+
83
+
84
+ def build_semantic_chunking(
85
+ discoverer: Optional[BoundaryDiscoverer] = None,
86
+ summarizer: Optional[Summarizer] = None,
87
+ *,
88
+ model_id: Optional[str] = None,
89
+ retry_policy: Any = DEFAULT_RETRY,
90
+ ):
91
+ """Compile the `semantic_chunking` subgraph. `discoverer` / `summarizer` are injected (defaulting to the
92
+ live single-call discoverer + seam summarizer), matching `rlm_chunking.chunk`. `retry_policy` is the
93
+ discover node's policy (overridable for fast tests)."""
94
+
95
+ discoverer = discoverer if discoverer is not None else SingleCallBoundaryDiscoverer(model_id)
96
+ summarizer = summarizer if summarizer is not None else SeamSummarizer()
97
+ max_attempts = int(getattr(retry_policy, "max_attempts", 3))
98
+
99
+ def gate(state: SemanticChunkingState) -> SemanticChunkingState:
100
+ path = _manifest_path(state)
101
+ if path.exists(): # content-hash gate: reuse, no LLM
102
+ return {"manifest": ChunkManifest.model_validate_json(path.read_text(encoding="utf-8"))}
103
+ state["cache_dir"].mkdir(parents=True, exist_ok=True)
104
+ return {"document": load_document(state["parsed"])}
105
+
106
+ def discover(state: SemanticChunkingState, runtime: Runtime) -> SemanticChunkingState:
107
+ # node_attempt is 1-indexed; on the FINAL attempt a failure dead-letters the document (skip it, never
108
+ # lose the batch) instead of raising; earlier attempts re-raise so the RetryPolicy retries.
109
+ attempt = runtime.execution_info.node_attempt
110
+ with raw_llm_span("semantic_chunking.discover", model=str(model_id or "single-call")):
111
+ try:
112
+ spans = discoverer.discover(state["document"])
113
+ except Exception as exc: # noqa: BLE001
114
+ if attempt >= max_attempts:
115
+ return {"dead_letter": dead_letter(
116
+ "boundary_discovery_failed", source_doc_id=state["parsed"].source_doc_id, error=str(exc))}
117
+ raise TransientExtraction(str(exc)) from exc
118
+ return {"spans": spans}
119
+
120
+ def finalize(state: SemanticChunkingState) -> SemanticChunkingState:
121
+ document, spans = state["document"], state["spans"]
122
+ token_cap = state.get("token_cap", DEFAULT_TOKEN_CAP)
123
+ try:
124
+ _validate_partition(spans, len(document.texts))
125
+ texts = _finalize_chunks(document, spans, token_cap)
126
+ except BoundaryValidationError as exc:
127
+ return {"dead_letter": dead_letter(
128
+ "boundary_validation_failed", source_doc_id=state["parsed"].source_doc_id, error=str(exc))}
129
+ return {"texts": texts, "offsets": _chunk_offsets(texts)}
130
+
131
+ def summarize(state: SemanticChunkingState) -> SemanticChunkingState:
132
+ # Sync node using asyncio.run (matches rlm_chunking.chunk). asyncio.run copies the current contextvars
133
+ # context, so the OTel trace context is preserved; the to_thread summary fan-out is carried by
134
+ # GraphWright's threading instrumentation. Works under both invoke and ainvoke (a sync node runs in a
135
+ # worker thread with no live loop during ainvoke).
136
+ with business_span("semantic_chunking.summarize", chunk_count=len(state["texts"])):
137
+ summaries = asyncio.run(_summarize_all(
138
+ state["texts"], summarizer, state.get("max_concurrency", DEFAULT_SUMMARY_CONCURRENCY)))
139
+ return {"summaries": summaries}
140
+
141
+ def manifest(state: SemanticChunkingState) -> SemanticChunkingState:
142
+ parsed = state["parsed"]
143
+ texts, summaries, offsets = state["texts"], state["summaries"], state["offsets"]
144
+ token_cap = state.get("token_cap", DEFAULT_TOKEN_CAP)
145
+ chunks = [
146
+ Chunk(
147
+ chunk_id=ChunkId.of(parsed.source_doc_id, i, text).value,
148
+ chunk_index=i,
149
+ text=text,
150
+ summary=summary,
151
+ token_estimate=_estimate_tokens(text),
152
+ doc_start=offsets[i][0],
153
+ doc_end=offsets[i][1],
154
+ )
155
+ for i, (text, summary) in enumerate(zip(texts, summaries))
156
+ ]
157
+ _validate_boundaries(chunks, token_cap)
158
+ result = ChunkManifest(
159
+ source_doc_id=parsed.source_doc_id, content_hash=parsed.content_hash, token_cap=token_cap, chunks=chunks)
160
+ _manifest_path(state).write_text(result.model_dump_json(indent=2), encoding="utf-8")
161
+ return {"manifest": result}
162
+
163
+ g = StateGraph(SemanticChunkingState)
164
+ g.add_node("gate", gate)
165
+ g.add_node("discover", discover, retry_policy=retry_policy)
166
+ g.add_node("finalize", finalize)
167
+ g.add_node("summarize", summarize)
168
+ g.add_node("manifest", manifest)
169
+
170
+ g.add_edge(START, "gate")
171
+ g.add_conditional_edges("gate", lambda s: "end" if s.get("manifest") else "discover",
172
+ {"discover": "discover", "end": END})
173
+ g.add_conditional_edges("discover", lambda s: "end" if s.get("dead_letter") else "finalize",
174
+ {"finalize": "finalize", "end": END})
175
+ g.add_conditional_edges("finalize", lambda s: "end" if s.get("dead_letter") else "summarize",
176
+ {"summarize": "summarize", "end": END})
177
+ g.add_edge("summarize", "manifest")
178
+ g.add_edge("manifest", END)
179
+ return g.compile()
180
+
181
+
182
+ # (EP-CORE-1a/ADR-0118: register_semantic_chunking_subgraph removed — semantic_chunking is de-registered from ARD;
183
+ # it's a core helper now. The LangGraph runnable + ChunkManifest contract stay; they're just not ARD-catalogued.)