rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,999 @@
1
+ """LG-3d: `contract_ingestion_pipeline` -- the GENERIC ingestion pipeline as a composite LangGraph subgraph.
2
+
3
+ Source documents -> a populated, connected contract KG, composing the LG-1/LG-2 ingest subgraphs + the
4
+ ingest-side capabilities. The pipeline is corpus-AGNOSTIC; each per-document ingest runs:
5
+
6
+ START --> chunk [RetryPolicy] (semantic_chunking: text -> chunks)
7
+ v
8
+ segment (SHARED: segment + BATCHED LLM function-classify -> spans; ADR-0048)
9
+ / | \\ (three branches fan out in parallel from the shared spans)
10
+ extract index extract_graph (clause extraction | dense/sparse Span index | graph extraction)
11
+ _clauses _spans
12
+ \\ | /
13
+ resolve (entity_resolution: mentions -> canonical entities)
14
+ v
15
+ write (write_clause_kg + write_graph; span index already written) --> END
16
+ | (any stage fails)
17
+ +--> dead_letter --> END (one bad document never kills the corpus ingest; the span
18
+ index is best-effort -- its failure never dead-letters the doc)
19
+
20
+ **The corpus seam (the whole point).** A `CorpusAdapter` yields `SourceDocument`s -- the ONLY per-corpus code.
21
+ Adding a corpus = writing one adapter (`documents() -> SourceDocument{canonical id, text, optional metadata}`),
22
+ NEVER re-implementing the flow: `run_corpus_ingestion(XYZAdapter(), pipeline)`, not an `ingest_xyz()`. Parsing
23
+ is the adapter's job (PDF via docling, CUAD from JSON, ...), so the generic pipeline starts from text.
24
+ `run_corpus_ingestion` maps every document through the pipeline (collecting per-document results and
25
+ dead-letters). (The KG-7 `party_clause_linking`/PartyTo post-step was retired -- issue 0028 / ADR-0091 --
26
+ since party->clause is reached via CONTRACTS_WITH provenance + the contract-scoped clause KG.)
27
+
28
+ Every stage is dependency-injected so the graph is hermetically testable with stubs -- no live LLM/store.
29
+ `production_contract_ingestion_pipeline` wires the real capabilities. The `source_doc_id` on every
30
+ `SourceDocument` MUST come from `canonical_source_doc_id` (HYG-1), so every graph shares one id scheme and the
31
+ KG-7 link is a clean join.
32
+ """
33
+
34
+ from __future__ import annotations
35
+
36
+ import asyncio
37
+ from dataclasses import dataclass
38
+ from typing import Any, Callable, Iterable, Optional, Protocol, TypedDict, runtime_checkable
39
+
40
+ from langgraph.graph import END, START, StateGraph
41
+ from langgraph.runtime import Runtime
42
+ from pydantic import BaseModel
43
+
44
+ from rag_wright.capabilities.parsing import ParsedDocument
45
+ # EP-API-6b: the generic corpus-seam contract + docling-parse helpers moved to a DOMAIN-FREE module (so the engine
46
+ # parse API does not import this contract pipeline). Re-exported here, unchanged, for this module's own importers.
47
+ from rag_wright.capabilities.document_parse import ( # noqa: F401 (re-export)
48
+ _INGEST_PARSE_DEADLINE_S,
49
+ SourceDocument,
50
+ aparsed_source_document,
51
+ parsed_source_document,
52
+ )
53
+ from rag_wright.subgraphs.scaffold import DEFAULT_RETRY, business_span, dead_letter
54
+ from rag_wright.subgraphs.typed_clause_extraction import TransientExtraction
55
+
56
+
57
+ class IngestionReport(BaseModel):
58
+ """The composite's output: how many documents were ingested and which dead-lettered (with reasons)."""
59
+
60
+ documents_ingested: int
61
+ dead_lettered: list[dict]
62
+ party_links: int = 0 # DEPRECATED (issue 0028 / ADR-0091): the PartyTo layer was retired; always 0. Kept a
63
+ # release so a consumer reading this field does not break; slated for removal.
64
+ per_document: list[dict]
65
+ # PROD-3 lossless invariant (ADR-0050): documents written but INCOMPLETE (>=1 clause extraction failed after
66
+ # retries, OR >=1 span-index write failed -- 0006-C). Surfaced here so a partial is KNOWN at job completion,
67
+ # never discovered later by grepping logs. Each entry:
68
+ # {"source_doc_id": str,
69
+ # "failures": [{"kind": "clause"|"span", "span_id": str, "reason": str}, ...], # ENG-1: read THIS -- always
70
+ # # present, covers BOTH kinds
71
+ # "clause_failures": [...], # present only if a clause loss (back-compat)
72
+ # "span_failures": [...]} # present only if a span loss (back-compat)
73
+ # Integrators: key on `failures` (or on the doc being in `partial` at all). Reading only `clause_failures`
74
+ # SILENTLY misses a span-only loss -- the per-kind keys are optional, `failures` is not.
75
+ partial: list[dict] = []
76
+
77
+
78
+ @runtime_checkable
79
+ class CorpusAdapter(Protocol):
80
+ """The ONE per-corpus seam: yield the corpus's documents as `SourceDocument`s (parsing + canonical id +
81
+ any corpus metadata live here). Everything downstream is corpus-agnostic."""
82
+
83
+ def documents(self) -> Iterable[SourceDocument]: ...
84
+
85
+
86
+ # The injected per-document stage seams (each wraps a built subgraph / capability; stubbed in tests).
87
+ ChunkFn = Callable[[SourceDocument], list] # doc -> chunks
88
+ SegmentFn = Callable[[SourceDocument, list], list] # (doc, chunks) -> [(op, primary_function, chunk_doc_start, scores)]
89
+ ClausesFn = Callable[[SourceDocument, list], Any] # (doc, segments) -> typed clause records (list) OR
90
+ # {clause_records, clause_failures} when the adapter reports per-clause failures (PROD-3 lossless; extract_clauses
91
+ # accepts either shape for back-compat)
92
+ IndexFn = Callable[[SourceDocument, list], int] # (doc, segments) -> #Span records written (retrieval index)
93
+ GraphFn = Callable[[SourceDocument, list], list] # (doc, chunks) -> ExtractionResults
94
+ ResolveFn = Callable[[list], Any] # extraction results -> resolution
95
+ WriteFn = Callable[[SourceDocument, list, Any], dict] # (doc, clause records, resolution) -> counts
96
+ LinkFn = Callable[[], int] # corpus-level post-ingest hook -> an int count (default no-op). The KG-7 PartyTo
97
+ # provider (`corpus_party_link_fn`) was retired (issue 0028 / ADR-0091); the seam
98
+ # stays for signature stability + a future corpus-level pass, default `lambda: 0`.
99
+
100
+
101
+ class IngestionState(TypedDict, total=False):
102
+ document: SourceDocument
103
+ chunks: list
104
+ segments: list # shared: [(OperativeSpan, primary_function, chunk_doc_start, [FunctionScore])] (ADR-0048)
105
+ clause_records: list
106
+ clause_failures: list # PROD-3 lossless: per-clause extraction failures (span_id + reason) -> doc flagged PARTIAL
107
+ span_count: int # Span records written to the retrieval index (best-effort)
108
+ span_failures: list # 0006-C (NFR-2): per-span index-write failures (span_id + reason) -> doc flagged PARTIAL
109
+ extraction_results: list
110
+ resolution: Any
111
+ written: dict
112
+ dead_letter: Optional[dict]
113
+
114
+
115
+
116
+
117
+ def _wire_ingest_graph(chunk, segment, extract_clauses, index_spans, extract_graph, resolve, write,
118
+ retry_policy) -> Any:
119
+ """The per-document ingest graph topology (shared by the sync `build_document_ingest` and the async
120
+ `abuild_document_ingest`). LangGraph `add_node` accepts sync OR async node functions, so the wiring is
121
+ identical -- only the node functions differ (sync `.invoke` vs async `.ainvoke`)."""
122
+ def _route(key: str):
123
+ return lambda state: "end" if state.get("dead_letter") else key
124
+
125
+ g = StateGraph(IngestionState)
126
+ g.add_node("chunk", chunk, retry_policy=retry_policy)
127
+ g.add_node("segment", segment, retry_policy=retry_policy)
128
+ g.add_node("extract_clauses", extract_clauses, retry_policy=retry_policy)
129
+ g.add_node("index_spans", index_spans)
130
+ g.add_node("extract_graph", extract_graph, retry_policy=retry_policy)
131
+ g.add_node("resolve", resolve)
132
+ g.add_node("write", write, retry_policy=retry_policy)
133
+
134
+ g.add_edge(START, "chunk")
135
+ g.add_conditional_edges("chunk", _route("segment"), {"segment": "segment", "end": END})
136
+ # after segmentation, fan out (parallel): clause extraction, the span index, and graph extraction
137
+ g.add_conditional_edges(
138
+ "segment", lambda s: "end" if s.get("dead_letter") else ["clauses", "index", "graph"],
139
+ {"clauses": "extract_clauses", "index": "index_spans", "graph": "extract_graph", "end": END})
140
+ g.add_edge("extract_clauses", "resolve") # resolve joins all three parallel branches
141
+ g.add_edge("index_spans", "resolve")
142
+ g.add_edge("extract_graph", "resolve")
143
+ g.add_edge("resolve", "write")
144
+ g.add_edge("write", END)
145
+ return g.compile()
146
+
147
+
148
+ def abuild_document_ingest(
149
+ chunk_fn: Any, segment_fn: Any, clauses_fn: Any, index_fn: Any, graph_fn: Any, resolve_fn: Any,
150
+ write_fn: Any, *, retry_policy: Any = DEFAULT_RETRY,
151
+ ):
152
+ """ASYNC-B2e (ADR-0057): the async per-document ingest subgraph. Same topology + dead-letter/retry semantics
153
+ as `build_document_ingest`, but the nodes are `async def` and the injected stage fns are awaited -- so the
154
+ model calls run on the async seam (true wall-clock deadline) and the parallel branches (clauses/index/graph)
155
+ run concurrently on the event loop. Invoke via `ainvoke`."""
156
+ max_attempts = int(getattr(retry_policy, "max_attempts", 3))
157
+
158
+ async def _aguard(name: str, work: Any, runtime: Runtime, doc: SourceDocument) -> dict:
159
+ attempt = runtime.execution_info.node_attempt
160
+ with business_span(f"contract_ingestion.{name}", source_doc_id=doc.source_doc_id):
161
+ try:
162
+ return await work()
163
+ except Exception as exc: # noqa: BLE001 - transient -> retry, or dead-letter on exhaustion
164
+ if attempt >= max_attempts:
165
+ return {"dead_letter": dead_letter(
166
+ "ingest_failed", source_doc_id=doc.source_doc_id, stage=name, error=str(exc))}
167
+ raise TransientExtraction(str(exc)) from exc
168
+
169
+ async def chunk(state: IngestionState, runtime: Runtime) -> IngestionState:
170
+ doc = state["document"]
171
+
172
+ async def _w() -> dict:
173
+ return {"chunks": await chunk_fn(doc)}
174
+
175
+ return await _aguard("chunk", _w, runtime, doc)
176
+
177
+ async def segment(state: IngestionState, runtime: Runtime) -> IngestionState:
178
+ doc = state["document"]
179
+
180
+ async def _w() -> dict:
181
+ return {"segments": await segment_fn(doc, state.get("chunks", []))}
182
+
183
+ return await _aguard("segment", _w, runtime, doc)
184
+
185
+ async def extract_clauses(state: IngestionState, runtime: Runtime) -> IngestionState:
186
+ doc = state["document"]
187
+
188
+ async def _w() -> dict:
189
+ result = await clauses_fn(doc, state.get("segments", []))
190
+ if isinstance(result, dict):
191
+ return {"clause_records": result.get("clause_records", []),
192
+ "clause_failures": result.get("clause_failures", [])}
193
+ return {"clause_records": result}
194
+
195
+ return await _aguard("extract_clauses", _w, runtime, doc)
196
+
197
+ async def index_spans(state: IngestionState) -> IngestionState:
198
+ if state.get("dead_letter"):
199
+ return {}
200
+ doc = state["document"]
201
+ with business_span("contract_ingestion.index_spans"):
202
+ try:
203
+ res = await index_fn(doc, state.get("segments", []))
204
+ except Exception as exc: # noqa: BLE001 - best-effort index: never dead-letters, but the loss is VISIBLE
205
+ return {"span_count": 0,
206
+ "span_failures": [{"span_id": "*", "reason": f"index node failed: {exc!r}"}]}
207
+ if isinstance(res, dict): # 0006-C: richer return surfaces per-span write failures (else a bare count)
208
+ return {"span_count": res.get("span_count", 0),
209
+ "span_failures": res.get("span_failures", [])}
210
+ return {"span_count": res}
211
+
212
+ async def extract_graph(state: IngestionState, runtime: Runtime) -> IngestionState:
213
+ doc = state["document"]
214
+
215
+ async def _w() -> dict:
216
+ return {"extraction_results": await graph_fn(doc, state.get("chunks", []))}
217
+
218
+ return await _aguard("extract_graph", _w, runtime, doc)
219
+
220
+ async def resolve(state: IngestionState) -> IngestionState:
221
+ if state.get("dead_letter"):
222
+ return {}
223
+ with business_span("contract_ingestion.resolve"):
224
+ return {"resolution": await resolve_fn(state.get("extraction_results", []))}
225
+
226
+ async def write(state: IngestionState, runtime: Runtime) -> IngestionState:
227
+ if state.get("dead_letter"):
228
+ return {}
229
+ doc = state["document"]
230
+
231
+ async def _do() -> dict:
232
+ counts = await write_fn(doc, state.get("clause_records", []), state.get("resolution"))
233
+ counts["spans"] = state.get("span_count", 0)
234
+ return {"written": counts}
235
+
236
+ return await _aguard("write", _do, runtime, doc)
237
+
238
+ return _wire_ingest_graph(chunk, segment, extract_clauses, index_spans, extract_graph, resolve, write,
239
+ retry_policy)
240
+
241
+
242
+ @dataclass
243
+ class PendingDocument:
244
+ """0009-ASYNC-INGEST: a document whose parse (incl. tiered OCR + the VLM escalation, the slowest call in the
245
+ pipeline) is DEFERRED. The async ingest parses it CONCURRENTLY and deadline-bounded PER DOCUMENT -- so a slow
246
+ scan on one document never blocks the loop or serializes the others -- instead of parsing every document
247
+ synchronously upfront. The GCS adapter yields these; a pre-parsed `SourceDocument` is used as-is."""
248
+
249
+ source_doc_id: str
250
+ parse: Callable[[], SourceDocument] # deferred: downloads + parses when called (run off-loop via to_thread)
251
+ metadata: dict
252
+
253
+
254
+
255
+
256
+ async def aparse_pending(pending: PendingDocument, *, deadline_s: float = _INGEST_PARSE_DEADLINE_S) -> SourceDocument:
257
+ """Run a `PendingDocument`'s deferred parse OFF the event loop (`to_thread`) under a wall-clock deadline
258
+ (ADR-0057), so the tiered OCR escalation is concurrency-safe and bounded during ingestion."""
259
+ async with asyncio.timeout(deadline_s):
260
+ sd = await asyncio.to_thread(pending.parse)
261
+ return sd.model_copy(update={"metadata": {**pending.metadata, **sd.metadata}})
262
+
263
+
264
+ def build_partial_entry(source_doc_id: str, clause_failures: list, span_failures: list,
265
+ ocr_failures: Optional[list] = None) -> Optional[dict]:
266
+ """The single PARTIAL-entry shape, shared by the blocking driver AND the async job runner so the two can
267
+ never drift. A UNIFIED, always-present `failures` list (kind-tagged) lets an integrator read ONE field and
268
+ never silently miss a span-only loss; the per-kind `clause_failures`/`span_failures` keys stay for
269
+ back-compat. Returns None when the document is fully complete (no loss -> not partial). Does not mutate the
270
+ input lists.
271
+
272
+ STABLE PUBLIC API (ENG-1/ENG-2): the product imports this helper and depends on its signature, this import
273
+ path, and the `failures` entry shape. Pinned by `tests/subgraphs/test_partial_entry_contract.py`. FORWARD-COMPAT
274
+ RULE: a new loss kind is a new `kind` value inside `failures` (0009-WIRE2 adds `ocr` -- a page a degraded scan
275
+ left unreadable), NEVER a replacement top-level key -- so an integrator counting the kind-tagged list keeps
276
+ surfacing losses it has no dedicated field for. `ocr_failures` is a new OPTIONAL trailing arg (3-arg callers
277
+ are unaffected)."""
278
+ clause_failures = clause_failures or []
279
+ span_failures = span_failures or []
280
+ ocr_failures = ocr_failures or []
281
+ if not (clause_failures or span_failures or ocr_failures):
282
+ return None
283
+ failures = ([{"kind": "clause", **f} for f in clause_failures]
284
+ + [{"kind": "span", **f} for f in span_failures]
285
+ + [{"kind": "ocr", **f} for f in ocr_failures])
286
+ entry: dict = {"source_doc_id": source_doc_id, "failures": failures}
287
+ if clause_failures:
288
+ entry["clause_failures"] = clause_failures
289
+ if span_failures:
290
+ entry["span_failures"] = span_failures
291
+ if ocr_failures:
292
+ entry["ocr_failures"] = ocr_failures
293
+ return entry
294
+
295
+
296
+ def _print_progress(message: str) -> None:
297
+ print(message, flush=True)
298
+
299
+
300
+
301
+
302
+ async def arun_corpus_ingestion(
303
+ adapter: CorpusAdapter, ingest_graph: Any, *, link_fn: LinkFn = lambda: 0,
304
+ progress: Callable[[str], None] = _print_progress,
305
+ is_done: Callable[[SourceDocument], bool] = lambda _doc: False,
306
+ job_id: str | None = None,
307
+ ) -> IngestionReport:
308
+ """ASYNC-B2e (ADR-0057): the async twin of `run_corpus_ingestion`. Maps each document through the ASYNC
309
+ per-document `ingest_graph` via `ainvoke` -- so the model calls carry the true wall-clock deadline and the
310
+ per-document parallel branches run concurrently on the loop. Same X/N progress, resume-skip, dead-letter, and
311
+ partial semantics as the sync driver (documents are processed sequentially; intra-document parallelism comes
312
+ from the graph).
313
+
314
+ Observability (issue 0017): each document's ingest runs inside `traced_run`, so EVERY generation it emits is
315
+ stamped with the correlation id -- `document_id = source_doc_id` and the caller's `job_id` -- making cost per
316
+ document (or per job) a single Langfuse query. A no-op unless RAG_TRACE_LEVEL is on + Langfuse configured."""
317
+ from rag_wright.models.tracing import traced_run
318
+ documents = list(adapter.documents())
319
+ total = len(documents)
320
+ progress(f"[ingest] starting: {total} documents")
321
+
322
+ ingested = skipped = 0
323
+ dead_lettered: list[dict] = []
324
+ partial: list[dict] = []
325
+ per_document: list[dict] = []
326
+ for i, item in enumerate(documents, 1):
327
+ if is_done(item):
328
+ ingested += 1
329
+ skipped += 1
330
+ if skipped % 25 == 0 or i == total:
331
+ progress(f"[ingest] {i}/{total} resume-skipping already-done docs ({skipped} skipped so far)")
332
+ continue
333
+ try: # 0009-ASYNC-INGEST: parse a deferred doc off-loop + deadline-bounded (no upfront-sync block)
334
+ document = await aparse_pending(item) if isinstance(item, PendingDocument) else item
335
+ except Exception as exc: # noqa: BLE001 - a parse-timeout/crash dead-letters THAT doc, never the run
336
+ dead_lettered.append({"source_doc_id": item.source_doc_id, "stage": "parse",
337
+ "reason": "parse_failed", "error": str(exc)[:200]})
338
+ progress(f"[ingest] {i}/{total} {item.source_doc_id} DEAD-LETTER (parse: {str(exc)[:80]})")
339
+ continue
340
+ with traced_run(document_id=document.source_doc_id, job_id=job_id, name="ingest_document"):
341
+ out = await ingest_graph.ainvoke({"document": document})
342
+ if out.get("dead_letter"):
343
+ dead_lettered.append(out["dead_letter"])
344
+ progress(f"[ingest] {i}/{total} {document.source_doc_id} DEAD-LETTER "
345
+ f"({out['dead_letter'].get('stage')}: {out['dead_letter'].get('reason')})")
346
+ continue
347
+ ingested += 1
348
+ written = out.get("written", {})
349
+ per_document.append({"source_doc_id": document.source_doc_id, "written": written})
350
+ summary = " ".join(f"{k}={v}" for k, v in written.items()) or "ok"
351
+ clause_failures = out.get("clause_failures") or []
352
+ span_failures = out.get("span_failures") or []
353
+ ocr_failures = [{"page": pg, "reason": "unreadable scan (OCR + VLM failed)"} # 0009-WIRE2
354
+ for pg in (getattr(document, "ocr_unreadable_pages", None) or [])]
355
+ entry = build_partial_entry(document.source_doc_id, clause_failures, span_failures, ocr_failures)
356
+ if entry is not None: # 0006-C / 0009: ANY kind of loss flags the doc PARTIAL (never silent)
357
+ partial.append(entry)
358
+ reasons = ", ".join(
359
+ p for p in (f"{len(clause_failures)} clause(s)" if clause_failures else "",
360
+ f"{len(span_failures)} span(s)" if span_failures else "",
361
+ f"{len(ocr_failures)} unreadable page(s)" if ocr_failures else "") if p)
362
+ progress(f"[ingest] {i}/{total} {document.source_doc_id} PARTIAL ({reasons} failed) {summary}")
363
+ else:
364
+ progress(f"[ingest] {i}/{total} {document.source_doc_id} OK {summary}")
365
+
366
+ party_links = link_fn() # default no-op (issue 0028: PartyTo retired); a caller may still pass a corpus-level hook
367
+ progress(f"[ingest] done: {ingested}/{total} ingested ({skipped} resume-skipped), "
368
+ f"{len(dead_lettered)} dead-lettered, {len(partial)} partial")
369
+ return IngestionReport(
370
+ documents_ingested=ingested, dead_lettered=dead_lettered,
371
+ party_links=party_links, per_document=per_document, partial=partial)
372
+
373
+
374
+ def _parsed_from_text(source_doc_id: str, text: str, parse_dir: Any):
375
+ """text -> a `ParsedDocument` (one TextItem per non-blank line), cached -- so the standard `chunk()` path
376
+ (which loads a real DoclingDocument) works from a text corpus. INGEST-REFACTOR: the shared version of the
377
+ per-script `_build_parsed`."""
378
+ import hashlib
379
+
380
+ from docling_core.types.doc.document import DoclingDocument
381
+ from docling_core.types.doc.labels import DocItemLabel
382
+
383
+ from rag_wright.capabilities.parsing import ParsedDocument
384
+
385
+ content_hash = hashlib.sha256(text.encode("utf-8")).hexdigest()
386
+ manifest_path = parse_dir / f"{source_doc_id}.{content_hash[:16]}.json"
387
+ if not manifest_path.exists():
388
+ doc = DoclingDocument(name=source_doc_id)
389
+ for line in text.split("\n"):
390
+ if line.strip():
391
+ doc.add_text(label=DocItemLabel.TEXT, text=line)
392
+ doc.save_as_json(manifest_path)
393
+ return ParsedDocument(source_doc_id=source_doc_id, content_hash=content_hash, manifest_path=str(manifest_path))
394
+
395
+
396
+ def _parsed_for(doc: SourceDocument, parse_dir: Any) -> ParsedDocument:
397
+ """CHUNK-7 (ADR-0058): the ParsedDocument the chunker chunks. Use the document's REAL docling parse
398
+ (`doc.parsed`, structure preserved) when a byte-source adapter provided one -- so the structural pass fires
399
+ on the document's own headings; otherwise fall back to a text-only parse of `doc.text` (genuinely
400
+ structureless input, which the chunker's tag-parse fallback handles). This is what carries docling structure
401
+ to the chunker."""
402
+ if doc.parsed is not None:
403
+ return doc.parsed
404
+ return _parsed_from_text(doc.source_doc_id, doc.text, parse_dir)
405
+
406
+
407
+ def _attach_page_provenance(doc: SourceDocument, chunks: list, segments: list, parse_dir: Any) -> list:
408
+ """issue 0032 (CU-B5): enrich each segment's `OperativeSpan` with its source page(s) + best-effort bbox.
409
+
410
+ Builds a page<->char map once from the parsed document's per-item `prov` pages and the canonical text, then
411
+ looks up each span's canonical range (`chunk_doc_start + op.start/end`). Deterministic, no model call.
412
+ Best-effort by design: any failure, a parse with no page provenance (the text-only ingest leg), or a chunk
413
+ with no `doc_start` leaves the span's `pages` empty -- the honest 'no page' fallback, never a broken ingest."""
414
+ if not segments:
415
+ return segments
416
+ try:
417
+ from rag_wright.capabilities.parsing import load_document
418
+ from rag_wright.capabilities.rlm_chunking import canonical_document_text
419
+ from rag_wright.corpus.document_parser import content_items
420
+ from rag_wright.spans.page_map import build_page_offset_map, pages_for
421
+
422
+ page_map = build_page_offset_map(
423
+ content_items(load_document(_parsed_for(doc, parse_dir))), canonical_document_text(chunks))
424
+ except Exception: # noqa: BLE001 - provenance is best-effort; never fail an ingest over a page lookup
425
+ return segments
426
+ if not page_map:
427
+ return segments
428
+ enriched: list = []
429
+ for (op, function, chunk_doc_start, scores) in segments:
430
+ if chunk_doc_start is None:
431
+ enriched.append((op, function, chunk_doc_start, scores))
432
+ continue
433
+ pages, bbox = pages_for(page_map, chunk_doc_start + op.start, chunk_doc_start + op.end)
434
+ enriched.append((op.model_copy(update={"pages": pages, "bbox": bbox}), function, chunk_doc_start, scores))
435
+ return enriched
436
+
437
+
438
+ class _NoSummary:
439
+ """A no-op summarizer -- the ingest smoke targets the typed KG + entity graph, not chunk summaries."""
440
+
441
+ def summarize(self, text: str) -> str: # noqa: ARG002
442
+ return ""
443
+
444
+
445
+ def seed_party_cache(party_dir: Any, legacy_path: Any) -> int:
446
+ """INGEST-REFACTOR (a): pre-populate the per-contract party cache from GP-1B's `dg_extracted_parties.json`
447
+ (a `{raw_title: [party names]}` map) so a full ingest REUSES those ~482 extractions instead of re-calling
448
+ granite. The legacy key is the raw title; it is canonicalized (HYG-1) to match the pipeline's
449
+ `source_doc_id`. Idempotent: never overwrites an existing (possibly fresher) entry. Returns #seeded."""
450
+ import json
451
+ from pathlib import Path
452
+
453
+ from rag_wright.contracts.identifiers import canonical_source_doc_id
454
+
455
+ if not Path(legacy_path).exists():
456
+ return 0
457
+ legacy = json.loads(Path(legacy_path).read_text(encoding="utf-8"))
458
+ Path(party_dir).mkdir(parents=True, exist_ok=True)
459
+ seeded = 0
460
+ for raw_title, names in legacy.items():
461
+ cache_file = Path(party_dir) / f"{canonical_source_doc_id(raw_title)}.json"
462
+ if not cache_file.exists():
463
+ cache_file.write_text(json.dumps(names), encoding="utf-8")
464
+ seeded += 1
465
+ return seeded
466
+
467
+
468
+ def seed_chunk_cache(chunk_dir: Any, legacy_chunk_dir: Any) -> int:
469
+ """INGEST-REFACTOR (a): copy existing `chunk()` manifests into the run's chunk cache so a full ingest skips
470
+ re-chunking already-chunked documents. The manifest name embeds the content hash, so a copied manifest is
471
+ only ever REUSED when the pipeline's text hashes to the same key (a text change misses, as it must).
472
+ Idempotent (skips existing). Returns #copied."""
473
+ import shutil
474
+ from pathlib import Path
475
+
476
+ if not Path(legacy_chunk_dir).exists():
477
+ return 0
478
+ Path(chunk_dir).mkdir(parents=True, exist_ok=True)
479
+ copied = 0
480
+ for manifest in Path(legacy_chunk_dir).glob("*.chunks.json"):
481
+ dest = Path(chunk_dir) / manifest.name
482
+ if not dest.exists():
483
+ shutil.copyfile(manifest, dest)
484
+ copied += 1
485
+ return copied
486
+
487
+
488
+ def per_contract_graph_extraction(doc: SourceDocument, *, party_dir: Any, names_fn: Callable[[str], list]) -> list:
489
+ """INGEST-REFACTOR (a) / GP-1B (ADR-0035): extract the signing parties ONCE PER CONTRACT, not per chunk.
490
+ Parties are named once in the preamble (the extractor bounds to `_DEFAULT_PREAMBLE_CHARS`), so one call per
491
+ contract is both the proven-fidelity design AND ~10x cheaper than the former per-chunk fan-out. Reuses cached
492
+ party names (seeded from `dg_extracted_parties.json` or a prior run) when present, else calls `names_fn` and
493
+ caches the result. `parties_to_extraction` rebuilds the exact `ExtractionResult` the live extractor would
494
+ (its own body is `names = [p.name for p in parties]; parties_to_extraction(...)`), so the cache is lossless.
495
+ Returns `[ExtractionResult]` (empty when no parties)."""
496
+ cache_file = _party_cache_file(party_dir, doc)
497
+ names = _cached_party_names(cache_file)
498
+ if names is None: # not cached -> extract once, then cache (empty results are cached too, as before)
499
+ names = names_fn(doc.text)
500
+ _write_party_cache(cache_file, names)
501
+ return _parties_extraction(names, doc)
502
+
503
+
504
+ async def aper_contract_graph_extraction(
505
+ doc: SourceDocument, *, party_dir: Any, anames_fn: Callable[[str], Any],
506
+ affil_dir: Any = None, aaffiliations_fn: Optional[Callable[[str], Any]] = None) -> list:
507
+ """ASYNC-B2c (ADR-0057): the async twin of `per_contract_graph_extraction`. The party-names model call runs on
508
+ the async seam via `anames_fn` (true wall-clock deadline); the cache and `parties_to_extraction` are sync.
509
+
510
+ issue 0027: when `aaffiliations_fn` + `affil_dir` are wired, ALSO extract corporate affiliations from the same
511
+ preamble (its own model call, separately cached, lexically pre-filtered), appending `AFFILIATE_OF` facts. Off
512
+ (params None) -> parties only, unchanged."""
513
+ cache_file = _party_cache_file(party_dir, doc)
514
+ names = _cached_party_names(cache_file)
515
+ if names is None:
516
+ names = await anames_fn(doc.text)
517
+ _write_party_cache(cache_file, names)
518
+ results = _parties_extraction(names, doc)
519
+ if aaffiliations_fn is not None and affil_dir is not None:
520
+ results = results + await _aaffiliations_extraction(doc, affil_dir, aaffiliations_fn)
521
+ return results
522
+
523
+
524
+ async def _aaffiliations_extraction(doc: SourceDocument, affil_dir: Any, aaffiliations_fn: Callable[[str], Any]) -> list:
525
+ """issue 0027: extract (or reuse cached) corporate-affiliation pairs for a contract, and rebuild the
526
+ `AFFILIATE_OF` ExtractionResult. Separate cache from parties (own file), so re-ingest never re-extracts."""
527
+ cache_file = _affil_cache_file(affil_dir, doc)
528
+ affiliations = _cached_json(cache_file)
529
+ if affiliations is None:
530
+ affiliations = list(await aaffiliations_fn(doc.text))
531
+ _write_json(cache_file, affiliations)
532
+ if not affiliations:
533
+ return []
534
+ from rag_wright.capabilities.graph_extraction import affiliations_to_extraction
535
+ from rag_wright.contracts.identifiers import ChunkId
536
+
537
+ return [affiliations_to_extraction(ChunkId.of(doc.source_doc_id, 0, doc.text), affiliations)]
538
+
539
+
540
+ def _affil_cache_file(affil_dir: Any, doc: SourceDocument) -> Any:
541
+ from pathlib import Path
542
+
543
+ return Path(affil_dir) / f"{doc.source_doc_id}.json"
544
+
545
+
546
+ def _cached_json(cache_file: Any) -> Optional[list]:
547
+ """The cached value (any JSON list) for a contract, or None when there is no cache entry (distinct from a
548
+ cached EMPTY result `[]`). Shared by the party and affiliation caches."""
549
+ import json
550
+
551
+ return json.loads(cache_file.read_text(encoding="utf-8")) if cache_file.exists() else None
552
+
553
+
554
+ def _write_json(cache_file: Any, value: list) -> None:
555
+ import json
556
+
557
+ cache_file.parent.mkdir(parents=True, exist_ok=True)
558
+ cache_file.write_text(json.dumps(value), encoding="utf-8")
559
+
560
+
561
+ def _party_cache_file(party_dir: Any, doc: SourceDocument) -> Any:
562
+ from pathlib import Path
563
+
564
+ return Path(party_dir) / f"{doc.source_doc_id}.json"
565
+
566
+
567
+ def _cached_party_names(cache_file: Any) -> Optional[list]:
568
+ """The cached party names for a contract, or None when there is no cache entry (distinct from a cached
569
+ EMPTY result, which is `[]`)."""
570
+ import json
571
+
572
+ return json.loads(cache_file.read_text(encoding="utf-8")) if cache_file.exists() else None
573
+
574
+
575
+ def _write_party_cache(cache_file: Any, names: list) -> None:
576
+ import json
577
+
578
+ cache_file.parent.mkdir(parents=True, exist_ok=True)
579
+ cache_file.write_text(json.dumps(names), encoding="utf-8")
580
+
581
+
582
+ def _parties_extraction(names: list, doc: SourceDocument) -> list:
583
+ if not names:
584
+ return []
585
+ from rag_wright.capabilities.graph_extraction import parties_to_extraction
586
+ from rag_wright.contracts.identifiers import ChunkId
587
+
588
+ return [parties_to_extraction(ChunkId.of(doc.source_doc_id, 0, doc.text), names)]
589
+
590
+
591
+
592
+
593
+ async def _asegment_and_classify(chunks: list, classify_fn: Any, *, segment: Any = None,
594
+ max_concurrency: int | None = None) -> list:
595
+ """ASYNC-B2e (ADR-0057): the async twin of `_segment_and_classify` -- classify each chunk's spans via the
596
+ async classifier (`aclassify_spans`, true wall-clock deadline). Segmentation (`segment_clause`) is CPU/regex,
597
+ kept sync.
598
+
599
+ CLASSIFY-CONCURRENCY-1: chunks are classified CONCURRENTLY (`asyncio.gather`), not one-after-another -- the
600
+ earlier `for ch: await ...` serialized M chunks into M network round-trips for no reason (nothing depends on
601
+ chunk order). ONE shared semaphore, threaded into every `aclassify_spans`, bounds the TOTAL in-flight
602
+ sub-batch LLM calls across all chunks to a single deliberate knob (`CLASSIFY_CONCURRENCY`, default 8) -- it is
603
+ acquired only at the leaf call, so the outer gather cannot deadlock. `gather` preserves order, so the flattened
604
+ output is identical to the sequential version, only faster."""
605
+ import os
606
+
607
+ from rag_wright.contracts.function import NO_FUNCTION, primary_function
608
+
609
+ seg = segment
610
+ if seg is None:
611
+ from rag_wright.spans.segment import segment_clause
612
+
613
+ seg = segment_clause
614
+ # segment (CPU/regex, sync) -> per-chunk operative spans, dropping empty chunks
615
+ per_chunk = [(ch, ops) for ch in chunks
616
+ if (ops := [op for op in seg(ch.chunk_id, ch.text) if op.text.strip()])]
617
+ if not per_chunk:
618
+ return []
619
+ n = max_concurrency if max_concurrency is not None else int(os.environ.get("CLASSIFY_CONCURRENCY", "8"))
620
+ sem = asyncio.Semaphore(n) # ONE shared bound on total in-flight classify calls (leaf-acquired -> no deadlock)
621
+ scores_by_chunk = await asyncio.gather(
622
+ *(classify_fn.aclassify_spans(ch.text, [op.text for op in ops], sem=sem) for ch, ops in per_chunk))
623
+ out: list = []
624
+ for (ch, ops), scores_per_span in zip(per_chunk, scores_by_chunk): # gather preserves order -> stable output
625
+ for op, scores in zip(ops, scores_per_span):
626
+ out.append((op, primary_function(scores) or NO_FUNCTION, ch.doc_start, scores))
627
+ return out
628
+
629
+
630
+ def clause_extraction_jobs(segments: list, boundary_starts: list[bool] | None = None) -> list:
631
+ """Group ordered `segments` into PROVISIONS and emit one clause-extraction job per provision, as
632
+ `[(index, anchor_op, function, scores, text)]`: `text` is the provision's merged span text (what the extractor
633
+ reads), `anchor_op` is its first span (the citation anchor + provenance), `index` is the provision ordinal.
634
+
635
+ Issue 0038: a `Clause` is a PROVISION, not a sentence. Extracting per span made 98% of spans clauses (a clause
636
+ per sentence) once issue 0036 removed the function gate -- the gate had been doing accidental provision
637
+ detection. The provision unit is the numbered section (`spans.segment.starts_new_provision`); retrieval stays
638
+ per span (`index_fn` is unchanged). A provision boundary is a CHUNK change OR a heading span, so granularity
639
+ self-adjusts: numbered sections -> provision-level; a heading-less document -> chunk-level (never per sentence,
640
+ never one clause per document).
641
+
642
+ Within a provision, `is_extractable_span` still drops furniture spans (page numbers, signature/notice labels)
643
+ from the merged text; a provision that is ALL furniture yields no clause. The FUNCTION stays a soft tag
644
+ (ADR-0082, issue 0036): the provision's function is the first non-NONE among its spans, else NO_FUNCTION -- an
645
+ untagged provision is still extracted (function-independent extraction).
646
+ """
647
+ from rag_wright.contracts.function import NO_FUNCTION
648
+ from rag_wright.spans.segment import is_extractable_span, starts_new_provision
649
+
650
+ # 1) group consecutive segments into provisions (boundary = chunk change OR a provision-heading span)
651
+ # `boundary_starts` (when provided) is the per-span "starts a new provision?" decision from
652
+ # `spans.boundary.adecide_provision_starts` (deterministic + the Jev residue fallback); None -> deterministic only.
653
+ groups: list[list] = []
654
+ for i, seg in enumerate(segments):
655
+ op = seg[0]
656
+ starts = boundary_starts[i] if boundary_starts is not None else starts_new_provision(op.text)
657
+ if groups and op.parent_chunk_id == groups[-1][-1][0].parent_chunk_id and not starts:
658
+ groups[-1].append(seg)
659
+ else:
660
+ groups.append([seg])
661
+
662
+ # 2) one job per provision, over its EXTRACTABLE spans only (furniture dropped; all-furniture -> no clause)
663
+ jobs: list = []
664
+ index = 0
665
+ for group in groups:
666
+ members = [seg for seg in group if is_extractable_span(seg[0].text)]
667
+ if not members:
668
+ continue
669
+ anchor_op = members[0][0]
670
+ text = "\n".join(seg[0].text.strip() for seg in members)
671
+ function = next((fn for (_op, fn, _cds, _sc) in members if fn and fn != NO_FUNCTION), NO_FUNCTION)
672
+ scores = members[0][3]
673
+ jobs.append((index, anchor_op, function, scores, text))
674
+ index += 1
675
+ return jobs
676
+
677
+
678
+ async def _aextract_clause_with_retry(
679
+ extractor: Any, *, chunk_id: Any, function: str, text: str, span_id: str, attempts: int,
680
+ functions: tuple[str, ...] = ()
681
+ ) -> tuple[Any, str]:
682
+ """Extract one clause with bounded retries. PARTIAL-CAUSE-1: docling-graph's `ExtractionFailed` is raised on
683
+ ANY logged docling error -- not only a deterministic "No valid JSON", but also TRANSIENT blips (an LLM empty
684
+ response, a gleaning failure, a rate-limit, a timeout). So EVERY failure is retried (the transient is what the
685
+ retry recovers); an earlier EXTRACT-GUARD-1 attempt to skip retrying `ExtractionFailed` turned recoverable
686
+ blips into lost clauses. The furniture that used to hard-fail deterministically is filtered UPSTREAM by
687
+ `is_extractable_span`, so this loop no longer retry-storms on non-clauses. Returns `(record, "")` on success or
688
+ `(None, reason)` on persistent failure."""
689
+ reason = ""
690
+ for _attempt in range(attempts):
691
+ try:
692
+ record = await extractor.aextract(chunk_id=chunk_id, function=function, text=text, span_id=span_id,
693
+ functions=functions)
694
+ return record, ""
695
+ except Exception as exc: # noqa: BLE001 - retry any failure (ExtractionFailed captures transients too)
696
+ reason = str(exc)
697
+ return None, reason
698
+
699
+
700
+ def _resolve_ingest_knobs(*, classify_concurrency: Any, clause_concurrency: Any, affiliations: Any,
701
+ function_classifier: Any) -> tuple:
702
+ """EP-API-4a: resolve the four ingest knobs, `None` -> the engine default (env fallback, so a non-API caller is
703
+ unaffected), else the explicit config override. Returns `(classify_concurrency, clause_concurrency, affiliations,
704
+ function_classifier_kind)`. `classify_concurrency` is passed through as-is (the segment leaf falls back to env
705
+ when it is None), so the whole chain keeps one env default per knob."""
706
+ import os
707
+
708
+ return (
709
+ classify_concurrency,
710
+ clause_concurrency if clause_concurrency is not None else int(os.environ.get("CLAUSE_CONCURRENCY", "8")),
711
+ affiliations if affiliations is not None else (os.getenv("RAG_INGEST_AFFILIATIONS", "1") != "0"),
712
+ (function_classifier or os.getenv("RAG_FUNCTION_CLASSIFIER", "setfit")).lower(),
713
+ )
714
+
715
+
716
+ def aproduction_document_ingest(
717
+ store: Any, *, cache_dir: Any, registry: Any, embedder: Any = None, party_seed_path: Any = None,
718
+ classify_fn: Any = None, extract_model: Any = None, list_model: Any = None, samples: Any = None,
719
+ graph_extract_model: Any = None, judge_model: Any = None, chunk_model: Any = None,
720
+ classify_concurrency: Any = None, clause_concurrency: Any = None, affiliations: Any = None,
721
+ function_classifier: Any = None, embedding_profile: str = "bge-m3"):
722
+ """ASYNC-B2e (ADR-0057): the async twin of `production_document_ingest`. Wires the ASYNC stage seams (achunk,
723
+ aclassify_spans, clause_extractor.aextract, aper_contract_graph_extraction) so the ingest model calls run on
724
+ the async seam with the true wall-clock deadline; CPU/store work (embed, resolve, DB writes) runs off the loop
725
+ via `asyncio.to_thread`. Clause extraction is bounded-concurrent via `asyncio.gather` + a `Semaphore`. Returns
726
+ an ASYNC per-document graph -- drive it with `arun_corpus_ingestion`.
727
+
728
+ Model configuration (mirrors the query/compliance entrypoints -- a caller no longer has to reach for env
729
+ vars to change the ingest models):
730
+ - `extract_model`: the PRIMARY clause-property extraction model -- an `ExtractionModel` OR a bare model-id
731
+ string (wrapped via `default_extraction_model`). `None` keeps the backend default (granite, per
732
+ `RAG_SERVING`). This is the model that produces the typed clause properties.
733
+ - `list_model`: the SECOND model for the cross-model UNION on the LIST-bearing groups only (carve_out /
734
+ covered_subject / damage_type). granite and gemma under-enumerate DIFFERENT list items, so their union
735
+ is more complete than either alone; the second model is cost-scoped to list groups. A bare model-id
736
+ string, `"off"` to disable, or `None` for the default (gemma, `RAG_INGEST_LIST_MODEL`). If you override
737
+ `extract_model` (e.g. to qwen), set `list_model` deliberately -- the union's value depends on the two
738
+ models being complementary.
739
+ - `samples`: same-model multi-sample count for the list union (`None` -> env `RAG_INGEST_CLAUSE_SAMPLES`,
740
+ default 1).
741
+ - `graph_extract_model`: the model for BOTH party AND affiliation extraction (the GP-1B graph-extract
742
+ surface -- they share one model). A bare model-id string or an `ExtractionModel` (unwrapped to its id);
743
+ `None` -> the default (granite, `RAG_GRAPH_EXTRACT_MODEL`).
744
+ - `judge_model`: the ingest semantic-judge model (ADR-0040 Layer-3 gate) -- a model-id string or an
745
+ `ExtractionModel`; `None` -> `model_for(STRUCTURED_REASONING)`.
746
+ - `chunk_model`: the chunker's boundary-refinement model -- structural boundaries are deterministic (zero
747
+ calls); ONLY an over-cap section triggers a bounded per-section tag-parse call, and this is the model it
748
+ uses. A model-id string or an `ExtractionModel`; `None` -> `model_for(GENERAL)`.
749
+ EP-API-4a ingest knobs (each `None` -> the env/default, so existing callers are unaffected):
750
+ `classify_concurrency` (function-classify parallelism), `clause_concurrency` (clause-extraction parallelism),
751
+ `affiliations` (run affiliation extraction), `function_classifier` ("setfit" | "llm"). The engine API passes
752
+ these from `EngineConfig.options.ingest`.
753
+ Env vars remain the fallback for every knob, so existing callers are unaffected."""
754
+ import asyncio
755
+ import hashlib
756
+ import json
757
+ from pathlib import Path
758
+
759
+ from rag_wright.capabilities.disambiguation import disambiguate
760
+ from rag_wright.capabilities.embedding_profiles import build_ingest_embedder
761
+ from rag_wright.capabilities.entity_resolution import resolve_entities
762
+ from rag_wright.capabilities.graph_extraction import aproduction_extract_fn
763
+ from rag_wright.capabilities.graph_storage import to_graph
764
+ from rag_wright.capabilities.rlm_chunking import StructuralModelFallbackDiscoverer, achunk
765
+ from rag_wright.contracts.contract_meta import ContractRecord
766
+ from rag_wright.contracts.identifiers import ChunkId
767
+ from rag_wright.contracts.property import ClausePropertyRecord
768
+ from rag_wright.models.profiles import ModelRole, model_for
769
+ from rag_wright.ontology.clause_template import Clause
770
+ from rag_wright.capabilities.dg_extraction import default_extraction_model
771
+ from rag_wright.spans.clause_kg_extractor import classifier_property_extractor
772
+ from rag_wright.spans.model_capabilities import (
773
+ CapabilityFunctionClassifier,
774
+ capability_property_classifier_fn,
775
+ )
776
+ from rag_wright.spans.segment import to_span_record
777
+
778
+ parse_dir = Path(cache_dir) / "parsed"
779
+ chunk_dir = Path(cache_dir) / "chunks"
780
+ clause_cache_dir = Path(cache_dir) / "clause_extract"
781
+ party_dir = Path(cache_dir) / "graph_parties"
782
+ affil_dir = Path(cache_dir) / "graph_affiliations" # issue 0027: separate from the party cache
783
+ for directory in (parse_dir, chunk_dir, clause_cache_dir, party_dir, affil_dir):
784
+ directory.mkdir(parents=True, exist_ok=True)
785
+ if party_seed_path is not None:
786
+ seed_party_cache(party_dir, party_seed_path)
787
+ # ADR-0058 (issue 0004): structure-first default -- deterministic boundaries from docling labels where present
788
+ # (zero model calls), bounded per-section TAG-PARSE fallback for over-cap sections (never the server-side
789
+ # guided-decoding whole-doc call that ran away past the 180s deadline). NOTE: this ingest currently flattens
790
+ # to text (`_parsed_from_text`), so labels are absent here and only the tag-parse fallback fires; preserving
791
+ # docling structure through ingest (a follow-up) unlocks the full zero-model structural win.
792
+ # the chunker's boundary-refinement model: structural boundaries are deterministic (zero calls); only an
793
+ # OVER-CAP section triggers a bounded per-section tag-parse call, and this is the model it uses (issue 0033
794
+ # follow-up). None -> default (GENERAL role). A bare id or an ExtractionModel (unwrapped to its id).
795
+ chunk_model_id = getattr(chunk_model, "model", chunk_model)
796
+ discoverer = StructuralModelFallbackDiscoverer(chunk_model_id)
797
+ summarizer = _NoSummary()
798
+ from rag_wright.spans.semantic_judge import build_asemantic_judge_fn
799
+ # caller-configurable ingest models (else backend/env defaults). A bare id -> an ExtractionModel; for the
800
+ # graph/judge surfaces (which take a model-id string) an ExtractionModel is unwrapped to its `.model` id.
801
+ clause_model = extract_model
802
+ if isinstance(extract_model, str):
803
+ clause_model = default_extraction_model("clause-extract", extract_model)
804
+ graph_extract_id = getattr(graph_extract_model, "model", graph_extract_model) # None or a bare model-id
805
+ judge_id = getattr(judge_model, "model", judge_model) or model_for(ModelRole.STRUCTURED_REASONING)
806
+ # CLS-D (ADR-0115): Step-3a property extraction is the classifier-first path -- the 29-dim best-of-both fleet,
807
+ # ONE residual LLM call for the 7 numeric/open dims (`clause_model`). EP-RT-7: the classifier LANE is dispatched
808
+ # through the `clause_property_classification` CAPABILITY (the single production path), never a second hand-built
809
+ # fleet here; the residual LLM call + the ADR-0028/0040/Layer-3 judge gates compose around it (ClassifierPropertyExtractor).
810
+ clause_extractor = classifier_property_extractor(
811
+ classifier_fn=capability_property_classifier_fn(),
812
+ model_id=getattr(clause_model, "model", clause_model), # the residual 7-numeric structured call
813
+ asemantic_judge_fn=build_asemantic_judge_fn(judge_id))
814
+ # party AND affiliation extraction share the graph-extract model (GP-1B); one arg drives both
815
+ aextract_parties_fn = (aproduction_extract_fn(model_id=graph_extract_id) if graph_extract_id
816
+ else aproduction_extract_fn())
817
+ # EP-API-4a: resolve the ingest knobs ONCE (config override else env/default), then use the resolved values.
818
+ _classify_concurrency, clause_concurrency, _affiliations_on, _clf_kind = _resolve_ingest_knobs(
819
+ classify_concurrency=classify_concurrency, clause_concurrency=clause_concurrency,
820
+ affiliations=affiliations, function_classifier=function_classifier)
821
+ if classify_fn is None:
822
+ # T55/SETFIT-SEG-1: the clause-function classifier is a SOFT tag (ADR-0047), so its implementation swaps
823
+ # behind this seam with NO contract/API change. DEFAULT is now the in-process trained SetFit ensemble
824
+ # soft-tagger (ms/span, no LLM call -- the ingestion-latency lever). `function_classifier="llm"` (or env
825
+ # RAG_FUNCTION_CLASSIFIER=llm) reverts to the LLM tag-classifier; a passed-in `classify_fn` overrides all.
826
+ if _clf_kind == "setfit":
827
+ # EP-RT-7: the default clause-function classifier dispatches through the `clause_function_classification`
828
+ # CAPABILITY (the single production path) -- not a second hand-built SetFit instance. (RAG_FUNCTION_CLASSIFIER=llm
829
+ # or an injected classify_fn are explicit non-capability overrides.)
830
+ classify_fn = CapabilityFunctionClassifier()
831
+ else:
832
+ from rag_wright.spans.clause_function_classifier import production_batch_clause_classifier
833
+
834
+ classify_fn = production_batch_clause_classifier(model_for(ModelRole.FUNCTION_CLASSIFY))
835
+ embedder = embedder if embedder is not None else build_ingest_embedder(embedding_profile)
836
+ template_version = hashlib.sha256(
837
+ json.dumps(Clause.model_json_schema(), sort_keys=True).encode("utf-8")).hexdigest()[:12]
838
+ _CLAUSE_EXTRACT_ATTEMPTS = 3 # clause_concurrency resolved above (EP-API-4a)
839
+
840
+ async def _aparty_names(text: str) -> list:
841
+ parties = await aextract_parties_fn(text)
842
+ return [p.name for p in parties.parties] if parties is not None else []
843
+
844
+ # issue 0027: corporate-affiliation extraction (AFFILIATE_OF). Default ON -- the lexical pre-filter keeps it a
845
+ # no-op for contracts that state no affiliation; config `affiliations=False` (or RAG_INGEST_AFFILIATIONS=0)
846
+ # disables it entirely. `_affiliations_on` resolved above (EP-API-4a).
847
+ async def _aaffiliations(text: str) -> list:
848
+ from rag_wright.capabilities.graph_extraction import aextract_affiliations
849
+
850
+ if graph_extract_id: # same graph-extract model as party extraction (issue 0033 follow-up)
851
+ return await aextract_affiliations(text, model_id=graph_extract_id)
852
+ return await aextract_affiliations(text)
853
+
854
+ async def chunk_fn(doc: SourceDocument) -> list:
855
+ parsed = _parsed_for(doc, parse_dir) # CHUNK-7: real docling parse if provided, else a text-only parse
856
+ manifest = await achunk(parsed, summarizer=summarizer, cache_dir=chunk_dir, discoverer=discoverer)
857
+ return list(manifest.chunks)
858
+
859
+ async def segment_fn(doc: SourceDocument, chunks: list) -> list:
860
+ segments = await _asegment_and_classify(chunks, classify_fn, max_concurrency=_classify_concurrency)
861
+ # issue 0032 (CU-B5): attach source-page provenance to each span. Build a page<->char-offset map over the
862
+ # canonical text (from the parsed doc's per-item prov pages) and look up each span's [doc_start, doc_end).
863
+ # Deterministic, no model call; degrades to no pages when the parse carried no provenance (text-only leg).
864
+ return _attach_page_provenance(doc, chunks, segments, parse_dir)
865
+
866
+ async def clauses_fn(doc: SourceDocument, segments: list) -> dict:
867
+ # issue 0038: a Clause is a PROVISION -- clause_extraction_jobs groups spans into provisions (numbered
868
+ # section, else chunk) and yields one job per provision (merged text + anchor span). Retrieval stays per
869
+ # span (index_fn unchanged). The function is a soft tag (ADR-0082), never a gate (issue 0036).
870
+ # Boundaries: deterministic-first, with a Jev decision-model fallback for the UNCERTAIN residue only
871
+ # (spans.boundary) -- flexible on new heading styles, degrades to deterministic with no decision model.
872
+ from rag_wright.spans.boundary import adecide_provision_starts, jev_boundary_decider
873
+
874
+ boundary_starts = await adecide_provision_starts(
875
+ [seg[0].text for seg in segments], decider=jev_boundary_decider())
876
+ jobs = clause_extraction_jobs(segments, boundary_starts=boundary_starts)
877
+ if not jobs:
878
+ return {"clause_records": [], "clause_failures": []}
879
+ failures: list[dict] = []
880
+ sem = asyncio.Semaphore(clause_concurrency)
881
+
882
+ from rag_wright.contracts.function import NO_FUNCTION as _NO_FUNCTION
883
+
884
+ async def _extract(job: Any) -> Any:
885
+ index, anchor_op, function, scores, text = job # text = merged provision; anchor_op = citation anchor
886
+ # CLS-D soft-scoping: the anchor span's TOP-3 real function soft-tags scope the classifier lane (union
887
+ # of their dims) -- tolerant of the ~0.5 function accuracy, and kills the over-emission a classifier
888
+ # (which cannot abstain) causes when run unscoped. An untagged span -> () -> no scoping (every dim runs).
889
+ functions = tuple(dict.fromkeys(
890
+ s.function for s in (scores or []) if s.function and s.function != _NO_FUNCTION))[:3]
891
+ clause_cid = ChunkId.of(doc.source_doc_id, index, text)
892
+ cache_file = clause_cache_dir / (hashlib.sha256(
893
+ f"{clause_cid.value}|{function}|{template_version}".encode("utf-8")).hexdigest()[:32] + ".json")
894
+ if cache_file.exists(): # a prior SUCCESSFUL extraction -> reuse it, no granite re-call
895
+ record = ClausePropertyRecord.model_validate_json(cache_file.read_text(encoding="utf-8"))
896
+ else:
897
+ record, reason = await _aextract_clause_with_retry(
898
+ clause_extractor, chunk_id=clause_cid, function=function, text=text,
899
+ span_id=anchor_op.span_id, attempts=_CLAUSE_EXTRACT_ATTEMPTS, functions=functions)
900
+ if record is None: # persistent failure -> record it (PARTIAL), do NOT cache, do NOT silently drop
901
+ failures.append({"span_id": anchor_op.span_id, "function": function, "reason": reason[:200]})
902
+ return None
903
+ cache_file.write_text(record.model_dump_json(), encoding="utf-8")
904
+ return record.model_copy(update={"functions": scores})
905
+
906
+ async def _bounded(job: Any) -> Any:
907
+ async with sem: # backpressure (network-bound granite)
908
+ return await _extract(job)
909
+
910
+ results = [r for r in await asyncio.gather(*(_bounded(j) for j in jobs)) if r is not None]
911
+ return {"clause_records": results, "clause_failures": failures}
912
+
913
+ async def index_fn(doc: SourceDocument, segments: list) -> dict:
914
+ if not segments:
915
+ return {"span_count": 0, "span_failures": []}
916
+ dense_vecs, sparse_vecs = await asyncio.to_thread(
917
+ embedder.encode_batch, [op.text.strip() for op, _, _, _ in segments])
918
+
919
+ def _write_all() -> dict:
920
+ count = 0
921
+ failures: list[dict] = []
922
+ for (op, function, chunk_doc_start, scores), dense, sparse in zip(segments, dense_vecs, sparse_vecs):
923
+ try:
924
+ store.upsert_span(to_span_record(
925
+ op, contract_id=doc.source_doc_id, chunk_doc_start=chunk_doc_start,
926
+ dense_vector=list(dense), sparse_vector=sparse, function=function,
927
+ functions=[s.function for s in scores])) # T55: top-k soft tags (primary-first)
928
+ count += 1
929
+ except Exception as exc: # noqa: BLE001 - a per-span write must not sink the KG, but is NOT swallowed
930
+ failures.append({"span_id": op.span_id, "reason": repr(exc)}) # 0006-C: surfaced -> PARTIAL
931
+ return {"span_count": count, "span_failures": failures}
932
+
933
+ return await asyncio.to_thread(_write_all)
934
+
935
+ async def graph_fn(doc: SourceDocument, chunks: list) -> list: # noqa: ARG001 - GP-1B is per-CONTRACT
936
+ return await aper_contract_graph_extraction(
937
+ doc, party_dir=party_dir, anames_fn=_aparty_names,
938
+ affil_dir=(affil_dir if _affiliations_on else None),
939
+ aaffiliations_fn=(_aaffiliations if _affiliations_on else None))
940
+
941
+ async def resolve_fn(extraction_results: list) -> Any:
942
+ return await asyncio.to_thread(
943
+ lambda: to_graph(resolve_entities(
944
+ disambiguate(extraction_results), extraction_results, resolver=registry))) # registry IS an EntityResolver (DD-3)
945
+
946
+ async def write_fn(doc: SourceDocument, clause_records: list, resolution: Any) -> dict:
947
+ from rag_wright.capabilities.contract_kg_store import ContractKGStore # DD-1b: clause KG + contract meta
948
+
949
+ ckg = ContractKGStore(store)
950
+
951
+ def _write() -> dict:
952
+ for record in clause_records:
953
+ ckg.write_clause_kg(record)
954
+ nodes, edges = resolution
955
+ store.write_graph(nodes, edges)
956
+ ckg.upsert_contract(ContractRecord(
957
+ contract_id=doc.source_doc_id, name=doc.metadata.get("raw_title", ""),
958
+ source_doc_id=doc.source_doc_id,
959
+ content_hash=hashlib.sha256(doc.text.encode("utf-8")).hexdigest()))
960
+ return {"clauses": len(clause_records), "entities": len(nodes), "edges": len(edges)}
961
+
962
+ return await asyncio.to_thread(_write)
963
+
964
+ return abuild_document_ingest(
965
+ chunk_fn, segment_fn, clauses_fn, index_fn, graph_fn, resolve_fn, write_fn)
966
+
967
+
968
+ # (issue 0028 / ADR-0091: `corpus_party_link_fn` -- the KG-7 PartyTo link provider -- was retired with the
969
+ # PartyTo edge. `arun_corpus_ingestion(link_fn=...)` keeps its no-op default; there is no PartyTo provider.)
970
+
971
+
972
+ def register_contract_ingestion_pipeline(registry) -> None:
973
+ """LG-3d: register `contract_ingestion_pipeline` (composite subgraph; source docs -> populated contract KG)."""
974
+ registry.register(
975
+ "contract_ingestion_pipeline",
976
+ contract=IngestionReport,
977
+ kind="subgraph",
978
+ display_name="Contract ingestion pipeline (corpus -> populated, connected KG)",
979
+ )
980
+
981
+
982
+ async def ainvoke(resources, inputs: dict):
983
+ """EP-CORE-2 (ADR-0118): the capability invoke factory (impl_ref target) -- ingest ONE document through the
984
+ async per-document graph over the opaque handle. `inputs`: document (an api.source_document / parse_document
985
+ SourceDocument) + cache_dir. Ingest knobs + embedder come from EngineConfig.options/embeddings (EP-API-4a/4b);
986
+ the entity resolver is the generic closed-world default (DD-3)."""
987
+ from rag_wright.capabilities.dg_extraction import default_extraction_model
988
+ from rag_wright.models.profiles import ModelRole
989
+ from rag_wright.ontology.registry import EntityRegistry
990
+
991
+ opts = resources._config.options.ingest
992
+ graph = aproduction_document_ingest(
993
+ resources._store, cache_dir=inputs["cache_dir"], registry=EntityRegistry(),
994
+ extract_model=default_extraction_model(model=resources.model_id(ModelRole.STRUCTURED_REASONING)),
995
+ embedding_profile=resources._config.embeddings.get("text", "bge-m3"),
996
+ list_model=opts.list_model, samples=opts.clause_samples,
997
+ classify_concurrency=opts.classify_concurrency, clause_concurrency=opts.clause_concurrency,
998
+ affiliations=opts.affiliations, function_classifier=opts.function_classifier)
999
+ return await graph.ainvoke({"document": inputs["document"]})