rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,999 @@
|
|
|
1
|
+
"""LG-3d: `contract_ingestion_pipeline` -- the GENERIC ingestion pipeline as a composite LangGraph subgraph.
|
|
2
|
+
|
|
3
|
+
Source documents -> a populated, connected contract KG, composing the LG-1/LG-2 ingest subgraphs + the
|
|
4
|
+
ingest-side capabilities. The pipeline is corpus-AGNOSTIC; each per-document ingest runs:
|
|
5
|
+
|
|
6
|
+
START --> chunk [RetryPolicy] (semantic_chunking: text -> chunks)
|
|
7
|
+
v
|
|
8
|
+
segment (SHARED: segment + BATCHED LLM function-classify -> spans; ADR-0048)
|
|
9
|
+
/ | \\ (three branches fan out in parallel from the shared spans)
|
|
10
|
+
extract index extract_graph (clause extraction | dense/sparse Span index | graph extraction)
|
|
11
|
+
_clauses _spans
|
|
12
|
+
\\ | /
|
|
13
|
+
resolve (entity_resolution: mentions -> canonical entities)
|
|
14
|
+
v
|
|
15
|
+
write (write_clause_kg + write_graph; span index already written) --> END
|
|
16
|
+
| (any stage fails)
|
|
17
|
+
+--> dead_letter --> END (one bad document never kills the corpus ingest; the span
|
|
18
|
+
index is best-effort -- its failure never dead-letters the doc)
|
|
19
|
+
|
|
20
|
+
**The corpus seam (the whole point).** A `CorpusAdapter` yields `SourceDocument`s -- the ONLY per-corpus code.
|
|
21
|
+
Adding a corpus = writing one adapter (`documents() -> SourceDocument{canonical id, text, optional metadata}`),
|
|
22
|
+
NEVER re-implementing the flow: `run_corpus_ingestion(XYZAdapter(), pipeline)`, not an `ingest_xyz()`. Parsing
|
|
23
|
+
is the adapter's job (PDF via docling, CUAD from JSON, ...), so the generic pipeline starts from text.
|
|
24
|
+
`run_corpus_ingestion` maps every document through the pipeline (collecting per-document results and
|
|
25
|
+
dead-letters). (The KG-7 `party_clause_linking`/PartyTo post-step was retired -- issue 0028 / ADR-0091 --
|
|
26
|
+
since party->clause is reached via CONTRACTS_WITH provenance + the contract-scoped clause KG.)
|
|
27
|
+
|
|
28
|
+
Every stage is dependency-injected so the graph is hermetically testable with stubs -- no live LLM/store.
|
|
29
|
+
`production_contract_ingestion_pipeline` wires the real capabilities. The `source_doc_id` on every
|
|
30
|
+
`SourceDocument` MUST come from `canonical_source_doc_id` (HYG-1), so every graph shares one id scheme and the
|
|
31
|
+
KG-7 link is a clean join.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
from __future__ import annotations
|
|
35
|
+
|
|
36
|
+
import asyncio
|
|
37
|
+
from dataclasses import dataclass
|
|
38
|
+
from typing import Any, Callable, Iterable, Optional, Protocol, TypedDict, runtime_checkable
|
|
39
|
+
|
|
40
|
+
from langgraph.graph import END, START, StateGraph
|
|
41
|
+
from langgraph.runtime import Runtime
|
|
42
|
+
from pydantic import BaseModel
|
|
43
|
+
|
|
44
|
+
from rag_wright.capabilities.parsing import ParsedDocument
|
|
45
|
+
# EP-API-6b: the generic corpus-seam contract + docling-parse helpers moved to a DOMAIN-FREE module (so the engine
|
|
46
|
+
# parse API does not import this contract pipeline). Re-exported here, unchanged, for this module's own importers.
|
|
47
|
+
from rag_wright.capabilities.document_parse import ( # noqa: F401 (re-export)
|
|
48
|
+
_INGEST_PARSE_DEADLINE_S,
|
|
49
|
+
SourceDocument,
|
|
50
|
+
aparsed_source_document,
|
|
51
|
+
parsed_source_document,
|
|
52
|
+
)
|
|
53
|
+
from rag_wright.subgraphs.scaffold import DEFAULT_RETRY, business_span, dead_letter
|
|
54
|
+
from rag_wright.subgraphs.typed_clause_extraction import TransientExtraction
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class IngestionReport(BaseModel):
|
|
58
|
+
"""The composite's output: how many documents were ingested and which dead-lettered (with reasons)."""
|
|
59
|
+
|
|
60
|
+
documents_ingested: int
|
|
61
|
+
dead_lettered: list[dict]
|
|
62
|
+
party_links: int = 0 # DEPRECATED (issue 0028 / ADR-0091): the PartyTo layer was retired; always 0. Kept a
|
|
63
|
+
# release so a consumer reading this field does not break; slated for removal.
|
|
64
|
+
per_document: list[dict]
|
|
65
|
+
# PROD-3 lossless invariant (ADR-0050): documents written but INCOMPLETE (>=1 clause extraction failed after
|
|
66
|
+
# retries, OR >=1 span-index write failed -- 0006-C). Surfaced here so a partial is KNOWN at job completion,
|
|
67
|
+
# never discovered later by grepping logs. Each entry:
|
|
68
|
+
# {"source_doc_id": str,
|
|
69
|
+
# "failures": [{"kind": "clause"|"span", "span_id": str, "reason": str}, ...], # ENG-1: read THIS -- always
|
|
70
|
+
# # present, covers BOTH kinds
|
|
71
|
+
# "clause_failures": [...], # present only if a clause loss (back-compat)
|
|
72
|
+
# "span_failures": [...]} # present only if a span loss (back-compat)
|
|
73
|
+
# Integrators: key on `failures` (or on the doc being in `partial` at all). Reading only `clause_failures`
|
|
74
|
+
# SILENTLY misses a span-only loss -- the per-kind keys are optional, `failures` is not.
|
|
75
|
+
partial: list[dict] = []
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
@runtime_checkable
|
|
79
|
+
class CorpusAdapter(Protocol):
|
|
80
|
+
"""The ONE per-corpus seam: yield the corpus's documents as `SourceDocument`s (parsing + canonical id +
|
|
81
|
+
any corpus metadata live here). Everything downstream is corpus-agnostic."""
|
|
82
|
+
|
|
83
|
+
def documents(self) -> Iterable[SourceDocument]: ...
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
# The injected per-document stage seams (each wraps a built subgraph / capability; stubbed in tests).
|
|
87
|
+
ChunkFn = Callable[[SourceDocument], list] # doc -> chunks
|
|
88
|
+
SegmentFn = Callable[[SourceDocument, list], list] # (doc, chunks) -> [(op, primary_function, chunk_doc_start, scores)]
|
|
89
|
+
ClausesFn = Callable[[SourceDocument, list], Any] # (doc, segments) -> typed clause records (list) OR
|
|
90
|
+
# {clause_records, clause_failures} when the adapter reports per-clause failures (PROD-3 lossless; extract_clauses
|
|
91
|
+
# accepts either shape for back-compat)
|
|
92
|
+
IndexFn = Callable[[SourceDocument, list], int] # (doc, segments) -> #Span records written (retrieval index)
|
|
93
|
+
GraphFn = Callable[[SourceDocument, list], list] # (doc, chunks) -> ExtractionResults
|
|
94
|
+
ResolveFn = Callable[[list], Any] # extraction results -> resolution
|
|
95
|
+
WriteFn = Callable[[SourceDocument, list, Any], dict] # (doc, clause records, resolution) -> counts
|
|
96
|
+
LinkFn = Callable[[], int] # corpus-level post-ingest hook -> an int count (default no-op). The KG-7 PartyTo
|
|
97
|
+
# provider (`corpus_party_link_fn`) was retired (issue 0028 / ADR-0091); the seam
|
|
98
|
+
# stays for signature stability + a future corpus-level pass, default `lambda: 0`.
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
class IngestionState(TypedDict, total=False):
|
|
102
|
+
document: SourceDocument
|
|
103
|
+
chunks: list
|
|
104
|
+
segments: list # shared: [(OperativeSpan, primary_function, chunk_doc_start, [FunctionScore])] (ADR-0048)
|
|
105
|
+
clause_records: list
|
|
106
|
+
clause_failures: list # PROD-3 lossless: per-clause extraction failures (span_id + reason) -> doc flagged PARTIAL
|
|
107
|
+
span_count: int # Span records written to the retrieval index (best-effort)
|
|
108
|
+
span_failures: list # 0006-C (NFR-2): per-span index-write failures (span_id + reason) -> doc flagged PARTIAL
|
|
109
|
+
extraction_results: list
|
|
110
|
+
resolution: Any
|
|
111
|
+
written: dict
|
|
112
|
+
dead_letter: Optional[dict]
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _wire_ingest_graph(chunk, segment, extract_clauses, index_spans, extract_graph, resolve, write,
|
|
118
|
+
retry_policy) -> Any:
|
|
119
|
+
"""The per-document ingest graph topology (shared by the sync `build_document_ingest` and the async
|
|
120
|
+
`abuild_document_ingest`). LangGraph `add_node` accepts sync OR async node functions, so the wiring is
|
|
121
|
+
identical -- only the node functions differ (sync `.invoke` vs async `.ainvoke`)."""
|
|
122
|
+
def _route(key: str):
|
|
123
|
+
return lambda state: "end" if state.get("dead_letter") else key
|
|
124
|
+
|
|
125
|
+
g = StateGraph(IngestionState)
|
|
126
|
+
g.add_node("chunk", chunk, retry_policy=retry_policy)
|
|
127
|
+
g.add_node("segment", segment, retry_policy=retry_policy)
|
|
128
|
+
g.add_node("extract_clauses", extract_clauses, retry_policy=retry_policy)
|
|
129
|
+
g.add_node("index_spans", index_spans)
|
|
130
|
+
g.add_node("extract_graph", extract_graph, retry_policy=retry_policy)
|
|
131
|
+
g.add_node("resolve", resolve)
|
|
132
|
+
g.add_node("write", write, retry_policy=retry_policy)
|
|
133
|
+
|
|
134
|
+
g.add_edge(START, "chunk")
|
|
135
|
+
g.add_conditional_edges("chunk", _route("segment"), {"segment": "segment", "end": END})
|
|
136
|
+
# after segmentation, fan out (parallel): clause extraction, the span index, and graph extraction
|
|
137
|
+
g.add_conditional_edges(
|
|
138
|
+
"segment", lambda s: "end" if s.get("dead_letter") else ["clauses", "index", "graph"],
|
|
139
|
+
{"clauses": "extract_clauses", "index": "index_spans", "graph": "extract_graph", "end": END})
|
|
140
|
+
g.add_edge("extract_clauses", "resolve") # resolve joins all three parallel branches
|
|
141
|
+
g.add_edge("index_spans", "resolve")
|
|
142
|
+
g.add_edge("extract_graph", "resolve")
|
|
143
|
+
g.add_edge("resolve", "write")
|
|
144
|
+
g.add_edge("write", END)
|
|
145
|
+
return g.compile()
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def abuild_document_ingest(
|
|
149
|
+
chunk_fn: Any, segment_fn: Any, clauses_fn: Any, index_fn: Any, graph_fn: Any, resolve_fn: Any,
|
|
150
|
+
write_fn: Any, *, retry_policy: Any = DEFAULT_RETRY,
|
|
151
|
+
):
|
|
152
|
+
"""ASYNC-B2e (ADR-0057): the async per-document ingest subgraph. Same topology + dead-letter/retry semantics
|
|
153
|
+
as `build_document_ingest`, but the nodes are `async def` and the injected stage fns are awaited -- so the
|
|
154
|
+
model calls run on the async seam (true wall-clock deadline) and the parallel branches (clauses/index/graph)
|
|
155
|
+
run concurrently on the event loop. Invoke via `ainvoke`."""
|
|
156
|
+
max_attempts = int(getattr(retry_policy, "max_attempts", 3))
|
|
157
|
+
|
|
158
|
+
async def _aguard(name: str, work: Any, runtime: Runtime, doc: SourceDocument) -> dict:
|
|
159
|
+
attempt = runtime.execution_info.node_attempt
|
|
160
|
+
with business_span(f"contract_ingestion.{name}", source_doc_id=doc.source_doc_id):
|
|
161
|
+
try:
|
|
162
|
+
return await work()
|
|
163
|
+
except Exception as exc: # noqa: BLE001 - transient -> retry, or dead-letter on exhaustion
|
|
164
|
+
if attempt >= max_attempts:
|
|
165
|
+
return {"dead_letter": dead_letter(
|
|
166
|
+
"ingest_failed", source_doc_id=doc.source_doc_id, stage=name, error=str(exc))}
|
|
167
|
+
raise TransientExtraction(str(exc)) from exc
|
|
168
|
+
|
|
169
|
+
async def chunk(state: IngestionState, runtime: Runtime) -> IngestionState:
|
|
170
|
+
doc = state["document"]
|
|
171
|
+
|
|
172
|
+
async def _w() -> dict:
|
|
173
|
+
return {"chunks": await chunk_fn(doc)}
|
|
174
|
+
|
|
175
|
+
return await _aguard("chunk", _w, runtime, doc)
|
|
176
|
+
|
|
177
|
+
async def segment(state: IngestionState, runtime: Runtime) -> IngestionState:
|
|
178
|
+
doc = state["document"]
|
|
179
|
+
|
|
180
|
+
async def _w() -> dict:
|
|
181
|
+
return {"segments": await segment_fn(doc, state.get("chunks", []))}
|
|
182
|
+
|
|
183
|
+
return await _aguard("segment", _w, runtime, doc)
|
|
184
|
+
|
|
185
|
+
async def extract_clauses(state: IngestionState, runtime: Runtime) -> IngestionState:
|
|
186
|
+
doc = state["document"]
|
|
187
|
+
|
|
188
|
+
async def _w() -> dict:
|
|
189
|
+
result = await clauses_fn(doc, state.get("segments", []))
|
|
190
|
+
if isinstance(result, dict):
|
|
191
|
+
return {"clause_records": result.get("clause_records", []),
|
|
192
|
+
"clause_failures": result.get("clause_failures", [])}
|
|
193
|
+
return {"clause_records": result}
|
|
194
|
+
|
|
195
|
+
return await _aguard("extract_clauses", _w, runtime, doc)
|
|
196
|
+
|
|
197
|
+
async def index_spans(state: IngestionState) -> IngestionState:
|
|
198
|
+
if state.get("dead_letter"):
|
|
199
|
+
return {}
|
|
200
|
+
doc = state["document"]
|
|
201
|
+
with business_span("contract_ingestion.index_spans"):
|
|
202
|
+
try:
|
|
203
|
+
res = await index_fn(doc, state.get("segments", []))
|
|
204
|
+
except Exception as exc: # noqa: BLE001 - best-effort index: never dead-letters, but the loss is VISIBLE
|
|
205
|
+
return {"span_count": 0,
|
|
206
|
+
"span_failures": [{"span_id": "*", "reason": f"index node failed: {exc!r}"}]}
|
|
207
|
+
if isinstance(res, dict): # 0006-C: richer return surfaces per-span write failures (else a bare count)
|
|
208
|
+
return {"span_count": res.get("span_count", 0),
|
|
209
|
+
"span_failures": res.get("span_failures", [])}
|
|
210
|
+
return {"span_count": res}
|
|
211
|
+
|
|
212
|
+
async def extract_graph(state: IngestionState, runtime: Runtime) -> IngestionState:
|
|
213
|
+
doc = state["document"]
|
|
214
|
+
|
|
215
|
+
async def _w() -> dict:
|
|
216
|
+
return {"extraction_results": await graph_fn(doc, state.get("chunks", []))}
|
|
217
|
+
|
|
218
|
+
return await _aguard("extract_graph", _w, runtime, doc)
|
|
219
|
+
|
|
220
|
+
async def resolve(state: IngestionState) -> IngestionState:
|
|
221
|
+
if state.get("dead_letter"):
|
|
222
|
+
return {}
|
|
223
|
+
with business_span("contract_ingestion.resolve"):
|
|
224
|
+
return {"resolution": await resolve_fn(state.get("extraction_results", []))}
|
|
225
|
+
|
|
226
|
+
async def write(state: IngestionState, runtime: Runtime) -> IngestionState:
|
|
227
|
+
if state.get("dead_letter"):
|
|
228
|
+
return {}
|
|
229
|
+
doc = state["document"]
|
|
230
|
+
|
|
231
|
+
async def _do() -> dict:
|
|
232
|
+
counts = await write_fn(doc, state.get("clause_records", []), state.get("resolution"))
|
|
233
|
+
counts["spans"] = state.get("span_count", 0)
|
|
234
|
+
return {"written": counts}
|
|
235
|
+
|
|
236
|
+
return await _aguard("write", _do, runtime, doc)
|
|
237
|
+
|
|
238
|
+
return _wire_ingest_graph(chunk, segment, extract_clauses, index_spans, extract_graph, resolve, write,
|
|
239
|
+
retry_policy)
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
@dataclass
|
|
243
|
+
class PendingDocument:
|
|
244
|
+
"""0009-ASYNC-INGEST: a document whose parse (incl. tiered OCR + the VLM escalation, the slowest call in the
|
|
245
|
+
pipeline) is DEFERRED. The async ingest parses it CONCURRENTLY and deadline-bounded PER DOCUMENT -- so a slow
|
|
246
|
+
scan on one document never blocks the loop or serializes the others -- instead of parsing every document
|
|
247
|
+
synchronously upfront. The GCS adapter yields these; a pre-parsed `SourceDocument` is used as-is."""
|
|
248
|
+
|
|
249
|
+
source_doc_id: str
|
|
250
|
+
parse: Callable[[], SourceDocument] # deferred: downloads + parses when called (run off-loop via to_thread)
|
|
251
|
+
metadata: dict
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
async def aparse_pending(pending: PendingDocument, *, deadline_s: float = _INGEST_PARSE_DEADLINE_S) -> SourceDocument:
|
|
257
|
+
"""Run a `PendingDocument`'s deferred parse OFF the event loop (`to_thread`) under a wall-clock deadline
|
|
258
|
+
(ADR-0057), so the tiered OCR escalation is concurrency-safe and bounded during ingestion."""
|
|
259
|
+
async with asyncio.timeout(deadline_s):
|
|
260
|
+
sd = await asyncio.to_thread(pending.parse)
|
|
261
|
+
return sd.model_copy(update={"metadata": {**pending.metadata, **sd.metadata}})
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def build_partial_entry(source_doc_id: str, clause_failures: list, span_failures: list,
|
|
265
|
+
ocr_failures: Optional[list] = None) -> Optional[dict]:
|
|
266
|
+
"""The single PARTIAL-entry shape, shared by the blocking driver AND the async job runner so the two can
|
|
267
|
+
never drift. A UNIFIED, always-present `failures` list (kind-tagged) lets an integrator read ONE field and
|
|
268
|
+
never silently miss a span-only loss; the per-kind `clause_failures`/`span_failures` keys stay for
|
|
269
|
+
back-compat. Returns None when the document is fully complete (no loss -> not partial). Does not mutate the
|
|
270
|
+
input lists.
|
|
271
|
+
|
|
272
|
+
STABLE PUBLIC API (ENG-1/ENG-2): the product imports this helper and depends on its signature, this import
|
|
273
|
+
path, and the `failures` entry shape. Pinned by `tests/subgraphs/test_partial_entry_contract.py`. FORWARD-COMPAT
|
|
274
|
+
RULE: a new loss kind is a new `kind` value inside `failures` (0009-WIRE2 adds `ocr` -- a page a degraded scan
|
|
275
|
+
left unreadable), NEVER a replacement top-level key -- so an integrator counting the kind-tagged list keeps
|
|
276
|
+
surfacing losses it has no dedicated field for. `ocr_failures` is a new OPTIONAL trailing arg (3-arg callers
|
|
277
|
+
are unaffected)."""
|
|
278
|
+
clause_failures = clause_failures or []
|
|
279
|
+
span_failures = span_failures or []
|
|
280
|
+
ocr_failures = ocr_failures or []
|
|
281
|
+
if not (clause_failures or span_failures or ocr_failures):
|
|
282
|
+
return None
|
|
283
|
+
failures = ([{"kind": "clause", **f} for f in clause_failures]
|
|
284
|
+
+ [{"kind": "span", **f} for f in span_failures]
|
|
285
|
+
+ [{"kind": "ocr", **f} for f in ocr_failures])
|
|
286
|
+
entry: dict = {"source_doc_id": source_doc_id, "failures": failures}
|
|
287
|
+
if clause_failures:
|
|
288
|
+
entry["clause_failures"] = clause_failures
|
|
289
|
+
if span_failures:
|
|
290
|
+
entry["span_failures"] = span_failures
|
|
291
|
+
if ocr_failures:
|
|
292
|
+
entry["ocr_failures"] = ocr_failures
|
|
293
|
+
return entry
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def _print_progress(message: str) -> None:
|
|
297
|
+
print(message, flush=True)
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
async def arun_corpus_ingestion(
|
|
303
|
+
adapter: CorpusAdapter, ingest_graph: Any, *, link_fn: LinkFn = lambda: 0,
|
|
304
|
+
progress: Callable[[str], None] = _print_progress,
|
|
305
|
+
is_done: Callable[[SourceDocument], bool] = lambda _doc: False,
|
|
306
|
+
job_id: str | None = None,
|
|
307
|
+
) -> IngestionReport:
|
|
308
|
+
"""ASYNC-B2e (ADR-0057): the async twin of `run_corpus_ingestion`. Maps each document through the ASYNC
|
|
309
|
+
per-document `ingest_graph` via `ainvoke` -- so the model calls carry the true wall-clock deadline and the
|
|
310
|
+
per-document parallel branches run concurrently on the loop. Same X/N progress, resume-skip, dead-letter, and
|
|
311
|
+
partial semantics as the sync driver (documents are processed sequentially; intra-document parallelism comes
|
|
312
|
+
from the graph).
|
|
313
|
+
|
|
314
|
+
Observability (issue 0017): each document's ingest runs inside `traced_run`, so EVERY generation it emits is
|
|
315
|
+
stamped with the correlation id -- `document_id = source_doc_id` and the caller's `job_id` -- making cost per
|
|
316
|
+
document (or per job) a single Langfuse query. A no-op unless RAG_TRACE_LEVEL is on + Langfuse configured."""
|
|
317
|
+
from rag_wright.models.tracing import traced_run
|
|
318
|
+
documents = list(adapter.documents())
|
|
319
|
+
total = len(documents)
|
|
320
|
+
progress(f"[ingest] starting: {total} documents")
|
|
321
|
+
|
|
322
|
+
ingested = skipped = 0
|
|
323
|
+
dead_lettered: list[dict] = []
|
|
324
|
+
partial: list[dict] = []
|
|
325
|
+
per_document: list[dict] = []
|
|
326
|
+
for i, item in enumerate(documents, 1):
|
|
327
|
+
if is_done(item):
|
|
328
|
+
ingested += 1
|
|
329
|
+
skipped += 1
|
|
330
|
+
if skipped % 25 == 0 or i == total:
|
|
331
|
+
progress(f"[ingest] {i}/{total} resume-skipping already-done docs ({skipped} skipped so far)")
|
|
332
|
+
continue
|
|
333
|
+
try: # 0009-ASYNC-INGEST: parse a deferred doc off-loop + deadline-bounded (no upfront-sync block)
|
|
334
|
+
document = await aparse_pending(item) if isinstance(item, PendingDocument) else item
|
|
335
|
+
except Exception as exc: # noqa: BLE001 - a parse-timeout/crash dead-letters THAT doc, never the run
|
|
336
|
+
dead_lettered.append({"source_doc_id": item.source_doc_id, "stage": "parse",
|
|
337
|
+
"reason": "parse_failed", "error": str(exc)[:200]})
|
|
338
|
+
progress(f"[ingest] {i}/{total} {item.source_doc_id} DEAD-LETTER (parse: {str(exc)[:80]})")
|
|
339
|
+
continue
|
|
340
|
+
with traced_run(document_id=document.source_doc_id, job_id=job_id, name="ingest_document"):
|
|
341
|
+
out = await ingest_graph.ainvoke({"document": document})
|
|
342
|
+
if out.get("dead_letter"):
|
|
343
|
+
dead_lettered.append(out["dead_letter"])
|
|
344
|
+
progress(f"[ingest] {i}/{total} {document.source_doc_id} DEAD-LETTER "
|
|
345
|
+
f"({out['dead_letter'].get('stage')}: {out['dead_letter'].get('reason')})")
|
|
346
|
+
continue
|
|
347
|
+
ingested += 1
|
|
348
|
+
written = out.get("written", {})
|
|
349
|
+
per_document.append({"source_doc_id": document.source_doc_id, "written": written})
|
|
350
|
+
summary = " ".join(f"{k}={v}" for k, v in written.items()) or "ok"
|
|
351
|
+
clause_failures = out.get("clause_failures") or []
|
|
352
|
+
span_failures = out.get("span_failures") or []
|
|
353
|
+
ocr_failures = [{"page": pg, "reason": "unreadable scan (OCR + VLM failed)"} # 0009-WIRE2
|
|
354
|
+
for pg in (getattr(document, "ocr_unreadable_pages", None) or [])]
|
|
355
|
+
entry = build_partial_entry(document.source_doc_id, clause_failures, span_failures, ocr_failures)
|
|
356
|
+
if entry is not None: # 0006-C / 0009: ANY kind of loss flags the doc PARTIAL (never silent)
|
|
357
|
+
partial.append(entry)
|
|
358
|
+
reasons = ", ".join(
|
|
359
|
+
p for p in (f"{len(clause_failures)} clause(s)" if clause_failures else "",
|
|
360
|
+
f"{len(span_failures)} span(s)" if span_failures else "",
|
|
361
|
+
f"{len(ocr_failures)} unreadable page(s)" if ocr_failures else "") if p)
|
|
362
|
+
progress(f"[ingest] {i}/{total} {document.source_doc_id} PARTIAL ({reasons} failed) {summary}")
|
|
363
|
+
else:
|
|
364
|
+
progress(f"[ingest] {i}/{total} {document.source_doc_id} OK {summary}")
|
|
365
|
+
|
|
366
|
+
party_links = link_fn() # default no-op (issue 0028: PartyTo retired); a caller may still pass a corpus-level hook
|
|
367
|
+
progress(f"[ingest] done: {ingested}/{total} ingested ({skipped} resume-skipped), "
|
|
368
|
+
f"{len(dead_lettered)} dead-lettered, {len(partial)} partial")
|
|
369
|
+
return IngestionReport(
|
|
370
|
+
documents_ingested=ingested, dead_lettered=dead_lettered,
|
|
371
|
+
party_links=party_links, per_document=per_document, partial=partial)
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
def _parsed_from_text(source_doc_id: str, text: str, parse_dir: Any):
|
|
375
|
+
"""text -> a `ParsedDocument` (one TextItem per non-blank line), cached -- so the standard `chunk()` path
|
|
376
|
+
(which loads a real DoclingDocument) works from a text corpus. INGEST-REFACTOR: the shared version of the
|
|
377
|
+
per-script `_build_parsed`."""
|
|
378
|
+
import hashlib
|
|
379
|
+
|
|
380
|
+
from docling_core.types.doc.document import DoclingDocument
|
|
381
|
+
from docling_core.types.doc.labels import DocItemLabel
|
|
382
|
+
|
|
383
|
+
from rag_wright.capabilities.parsing import ParsedDocument
|
|
384
|
+
|
|
385
|
+
content_hash = hashlib.sha256(text.encode("utf-8")).hexdigest()
|
|
386
|
+
manifest_path = parse_dir / f"{source_doc_id}.{content_hash[:16]}.json"
|
|
387
|
+
if not manifest_path.exists():
|
|
388
|
+
doc = DoclingDocument(name=source_doc_id)
|
|
389
|
+
for line in text.split("\n"):
|
|
390
|
+
if line.strip():
|
|
391
|
+
doc.add_text(label=DocItemLabel.TEXT, text=line)
|
|
392
|
+
doc.save_as_json(manifest_path)
|
|
393
|
+
return ParsedDocument(source_doc_id=source_doc_id, content_hash=content_hash, manifest_path=str(manifest_path))
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
def _parsed_for(doc: SourceDocument, parse_dir: Any) -> ParsedDocument:
|
|
397
|
+
"""CHUNK-7 (ADR-0058): the ParsedDocument the chunker chunks. Use the document's REAL docling parse
|
|
398
|
+
(`doc.parsed`, structure preserved) when a byte-source adapter provided one -- so the structural pass fires
|
|
399
|
+
on the document's own headings; otherwise fall back to a text-only parse of `doc.text` (genuinely
|
|
400
|
+
structureless input, which the chunker's tag-parse fallback handles). This is what carries docling structure
|
|
401
|
+
to the chunker."""
|
|
402
|
+
if doc.parsed is not None:
|
|
403
|
+
return doc.parsed
|
|
404
|
+
return _parsed_from_text(doc.source_doc_id, doc.text, parse_dir)
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
def _attach_page_provenance(doc: SourceDocument, chunks: list, segments: list, parse_dir: Any) -> list:
|
|
408
|
+
"""issue 0032 (CU-B5): enrich each segment's `OperativeSpan` with its source page(s) + best-effort bbox.
|
|
409
|
+
|
|
410
|
+
Builds a page<->char map once from the parsed document's per-item `prov` pages and the canonical text, then
|
|
411
|
+
looks up each span's canonical range (`chunk_doc_start + op.start/end`). Deterministic, no model call.
|
|
412
|
+
Best-effort by design: any failure, a parse with no page provenance (the text-only ingest leg), or a chunk
|
|
413
|
+
with no `doc_start` leaves the span's `pages` empty -- the honest 'no page' fallback, never a broken ingest."""
|
|
414
|
+
if not segments:
|
|
415
|
+
return segments
|
|
416
|
+
try:
|
|
417
|
+
from rag_wright.capabilities.parsing import load_document
|
|
418
|
+
from rag_wright.capabilities.rlm_chunking import canonical_document_text
|
|
419
|
+
from rag_wright.corpus.document_parser import content_items
|
|
420
|
+
from rag_wright.spans.page_map import build_page_offset_map, pages_for
|
|
421
|
+
|
|
422
|
+
page_map = build_page_offset_map(
|
|
423
|
+
content_items(load_document(_parsed_for(doc, parse_dir))), canonical_document_text(chunks))
|
|
424
|
+
except Exception: # noqa: BLE001 - provenance is best-effort; never fail an ingest over a page lookup
|
|
425
|
+
return segments
|
|
426
|
+
if not page_map:
|
|
427
|
+
return segments
|
|
428
|
+
enriched: list = []
|
|
429
|
+
for (op, function, chunk_doc_start, scores) in segments:
|
|
430
|
+
if chunk_doc_start is None:
|
|
431
|
+
enriched.append((op, function, chunk_doc_start, scores))
|
|
432
|
+
continue
|
|
433
|
+
pages, bbox = pages_for(page_map, chunk_doc_start + op.start, chunk_doc_start + op.end)
|
|
434
|
+
enriched.append((op.model_copy(update={"pages": pages, "bbox": bbox}), function, chunk_doc_start, scores))
|
|
435
|
+
return enriched
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
class _NoSummary:
|
|
439
|
+
"""A no-op summarizer -- the ingest smoke targets the typed KG + entity graph, not chunk summaries."""
|
|
440
|
+
|
|
441
|
+
def summarize(self, text: str) -> str: # noqa: ARG002
|
|
442
|
+
return ""
|
|
443
|
+
|
|
444
|
+
|
|
445
|
+
def seed_party_cache(party_dir: Any, legacy_path: Any) -> int:
|
|
446
|
+
"""INGEST-REFACTOR (a): pre-populate the per-contract party cache from GP-1B's `dg_extracted_parties.json`
|
|
447
|
+
(a `{raw_title: [party names]}` map) so a full ingest REUSES those ~482 extractions instead of re-calling
|
|
448
|
+
granite. The legacy key is the raw title; it is canonicalized (HYG-1) to match the pipeline's
|
|
449
|
+
`source_doc_id`. Idempotent: never overwrites an existing (possibly fresher) entry. Returns #seeded."""
|
|
450
|
+
import json
|
|
451
|
+
from pathlib import Path
|
|
452
|
+
|
|
453
|
+
from rag_wright.contracts.identifiers import canonical_source_doc_id
|
|
454
|
+
|
|
455
|
+
if not Path(legacy_path).exists():
|
|
456
|
+
return 0
|
|
457
|
+
legacy = json.loads(Path(legacy_path).read_text(encoding="utf-8"))
|
|
458
|
+
Path(party_dir).mkdir(parents=True, exist_ok=True)
|
|
459
|
+
seeded = 0
|
|
460
|
+
for raw_title, names in legacy.items():
|
|
461
|
+
cache_file = Path(party_dir) / f"{canonical_source_doc_id(raw_title)}.json"
|
|
462
|
+
if not cache_file.exists():
|
|
463
|
+
cache_file.write_text(json.dumps(names), encoding="utf-8")
|
|
464
|
+
seeded += 1
|
|
465
|
+
return seeded
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
def seed_chunk_cache(chunk_dir: Any, legacy_chunk_dir: Any) -> int:
|
|
469
|
+
"""INGEST-REFACTOR (a): copy existing `chunk()` manifests into the run's chunk cache so a full ingest skips
|
|
470
|
+
re-chunking already-chunked documents. The manifest name embeds the content hash, so a copied manifest is
|
|
471
|
+
only ever REUSED when the pipeline's text hashes to the same key (a text change misses, as it must).
|
|
472
|
+
Idempotent (skips existing). Returns #copied."""
|
|
473
|
+
import shutil
|
|
474
|
+
from pathlib import Path
|
|
475
|
+
|
|
476
|
+
if not Path(legacy_chunk_dir).exists():
|
|
477
|
+
return 0
|
|
478
|
+
Path(chunk_dir).mkdir(parents=True, exist_ok=True)
|
|
479
|
+
copied = 0
|
|
480
|
+
for manifest in Path(legacy_chunk_dir).glob("*.chunks.json"):
|
|
481
|
+
dest = Path(chunk_dir) / manifest.name
|
|
482
|
+
if not dest.exists():
|
|
483
|
+
shutil.copyfile(manifest, dest)
|
|
484
|
+
copied += 1
|
|
485
|
+
return copied
|
|
486
|
+
|
|
487
|
+
|
|
488
|
+
def per_contract_graph_extraction(doc: SourceDocument, *, party_dir: Any, names_fn: Callable[[str], list]) -> list:
|
|
489
|
+
"""INGEST-REFACTOR (a) / GP-1B (ADR-0035): extract the signing parties ONCE PER CONTRACT, not per chunk.
|
|
490
|
+
Parties are named once in the preamble (the extractor bounds to `_DEFAULT_PREAMBLE_CHARS`), so one call per
|
|
491
|
+
contract is both the proven-fidelity design AND ~10x cheaper than the former per-chunk fan-out. Reuses cached
|
|
492
|
+
party names (seeded from `dg_extracted_parties.json` or a prior run) when present, else calls `names_fn` and
|
|
493
|
+
caches the result. `parties_to_extraction` rebuilds the exact `ExtractionResult` the live extractor would
|
|
494
|
+
(its own body is `names = [p.name for p in parties]; parties_to_extraction(...)`), so the cache is lossless.
|
|
495
|
+
Returns `[ExtractionResult]` (empty when no parties)."""
|
|
496
|
+
cache_file = _party_cache_file(party_dir, doc)
|
|
497
|
+
names = _cached_party_names(cache_file)
|
|
498
|
+
if names is None: # not cached -> extract once, then cache (empty results are cached too, as before)
|
|
499
|
+
names = names_fn(doc.text)
|
|
500
|
+
_write_party_cache(cache_file, names)
|
|
501
|
+
return _parties_extraction(names, doc)
|
|
502
|
+
|
|
503
|
+
|
|
504
|
+
async def aper_contract_graph_extraction(
|
|
505
|
+
doc: SourceDocument, *, party_dir: Any, anames_fn: Callable[[str], Any],
|
|
506
|
+
affil_dir: Any = None, aaffiliations_fn: Optional[Callable[[str], Any]] = None) -> list:
|
|
507
|
+
"""ASYNC-B2c (ADR-0057): the async twin of `per_contract_graph_extraction`. The party-names model call runs on
|
|
508
|
+
the async seam via `anames_fn` (true wall-clock deadline); the cache and `parties_to_extraction` are sync.
|
|
509
|
+
|
|
510
|
+
issue 0027: when `aaffiliations_fn` + `affil_dir` are wired, ALSO extract corporate affiliations from the same
|
|
511
|
+
preamble (its own model call, separately cached, lexically pre-filtered), appending `AFFILIATE_OF` facts. Off
|
|
512
|
+
(params None) -> parties only, unchanged."""
|
|
513
|
+
cache_file = _party_cache_file(party_dir, doc)
|
|
514
|
+
names = _cached_party_names(cache_file)
|
|
515
|
+
if names is None:
|
|
516
|
+
names = await anames_fn(doc.text)
|
|
517
|
+
_write_party_cache(cache_file, names)
|
|
518
|
+
results = _parties_extraction(names, doc)
|
|
519
|
+
if aaffiliations_fn is not None and affil_dir is not None:
|
|
520
|
+
results = results + await _aaffiliations_extraction(doc, affil_dir, aaffiliations_fn)
|
|
521
|
+
return results
|
|
522
|
+
|
|
523
|
+
|
|
524
|
+
async def _aaffiliations_extraction(doc: SourceDocument, affil_dir: Any, aaffiliations_fn: Callable[[str], Any]) -> list:
|
|
525
|
+
"""issue 0027: extract (or reuse cached) corporate-affiliation pairs for a contract, and rebuild the
|
|
526
|
+
`AFFILIATE_OF` ExtractionResult. Separate cache from parties (own file), so re-ingest never re-extracts."""
|
|
527
|
+
cache_file = _affil_cache_file(affil_dir, doc)
|
|
528
|
+
affiliations = _cached_json(cache_file)
|
|
529
|
+
if affiliations is None:
|
|
530
|
+
affiliations = list(await aaffiliations_fn(doc.text))
|
|
531
|
+
_write_json(cache_file, affiliations)
|
|
532
|
+
if not affiliations:
|
|
533
|
+
return []
|
|
534
|
+
from rag_wright.capabilities.graph_extraction import affiliations_to_extraction
|
|
535
|
+
from rag_wright.contracts.identifiers import ChunkId
|
|
536
|
+
|
|
537
|
+
return [affiliations_to_extraction(ChunkId.of(doc.source_doc_id, 0, doc.text), affiliations)]
|
|
538
|
+
|
|
539
|
+
|
|
540
|
+
def _affil_cache_file(affil_dir: Any, doc: SourceDocument) -> Any:
|
|
541
|
+
from pathlib import Path
|
|
542
|
+
|
|
543
|
+
return Path(affil_dir) / f"{doc.source_doc_id}.json"
|
|
544
|
+
|
|
545
|
+
|
|
546
|
+
def _cached_json(cache_file: Any) -> Optional[list]:
|
|
547
|
+
"""The cached value (any JSON list) for a contract, or None when there is no cache entry (distinct from a
|
|
548
|
+
cached EMPTY result `[]`). Shared by the party and affiliation caches."""
|
|
549
|
+
import json
|
|
550
|
+
|
|
551
|
+
return json.loads(cache_file.read_text(encoding="utf-8")) if cache_file.exists() else None
|
|
552
|
+
|
|
553
|
+
|
|
554
|
+
def _write_json(cache_file: Any, value: list) -> None:
|
|
555
|
+
import json
|
|
556
|
+
|
|
557
|
+
cache_file.parent.mkdir(parents=True, exist_ok=True)
|
|
558
|
+
cache_file.write_text(json.dumps(value), encoding="utf-8")
|
|
559
|
+
|
|
560
|
+
|
|
561
|
+
def _party_cache_file(party_dir: Any, doc: SourceDocument) -> Any:
|
|
562
|
+
from pathlib import Path
|
|
563
|
+
|
|
564
|
+
return Path(party_dir) / f"{doc.source_doc_id}.json"
|
|
565
|
+
|
|
566
|
+
|
|
567
|
+
def _cached_party_names(cache_file: Any) -> Optional[list]:
|
|
568
|
+
"""The cached party names for a contract, or None when there is no cache entry (distinct from a cached
|
|
569
|
+
EMPTY result, which is `[]`)."""
|
|
570
|
+
import json
|
|
571
|
+
|
|
572
|
+
return json.loads(cache_file.read_text(encoding="utf-8")) if cache_file.exists() else None
|
|
573
|
+
|
|
574
|
+
|
|
575
|
+
def _write_party_cache(cache_file: Any, names: list) -> None:
|
|
576
|
+
import json
|
|
577
|
+
|
|
578
|
+
cache_file.parent.mkdir(parents=True, exist_ok=True)
|
|
579
|
+
cache_file.write_text(json.dumps(names), encoding="utf-8")
|
|
580
|
+
|
|
581
|
+
|
|
582
|
+
def _parties_extraction(names: list, doc: SourceDocument) -> list:
|
|
583
|
+
if not names:
|
|
584
|
+
return []
|
|
585
|
+
from rag_wright.capabilities.graph_extraction import parties_to_extraction
|
|
586
|
+
from rag_wright.contracts.identifiers import ChunkId
|
|
587
|
+
|
|
588
|
+
return [parties_to_extraction(ChunkId.of(doc.source_doc_id, 0, doc.text), names)]
|
|
589
|
+
|
|
590
|
+
|
|
591
|
+
|
|
592
|
+
|
|
593
|
+
async def _asegment_and_classify(chunks: list, classify_fn: Any, *, segment: Any = None,
|
|
594
|
+
max_concurrency: int | None = None) -> list:
|
|
595
|
+
"""ASYNC-B2e (ADR-0057): the async twin of `_segment_and_classify` -- classify each chunk's spans via the
|
|
596
|
+
async classifier (`aclassify_spans`, true wall-clock deadline). Segmentation (`segment_clause`) is CPU/regex,
|
|
597
|
+
kept sync.
|
|
598
|
+
|
|
599
|
+
CLASSIFY-CONCURRENCY-1: chunks are classified CONCURRENTLY (`asyncio.gather`), not one-after-another -- the
|
|
600
|
+
earlier `for ch: await ...` serialized M chunks into M network round-trips for no reason (nothing depends on
|
|
601
|
+
chunk order). ONE shared semaphore, threaded into every `aclassify_spans`, bounds the TOTAL in-flight
|
|
602
|
+
sub-batch LLM calls across all chunks to a single deliberate knob (`CLASSIFY_CONCURRENCY`, default 8) -- it is
|
|
603
|
+
acquired only at the leaf call, so the outer gather cannot deadlock. `gather` preserves order, so the flattened
|
|
604
|
+
output is identical to the sequential version, only faster."""
|
|
605
|
+
import os
|
|
606
|
+
|
|
607
|
+
from rag_wright.contracts.function import NO_FUNCTION, primary_function
|
|
608
|
+
|
|
609
|
+
seg = segment
|
|
610
|
+
if seg is None:
|
|
611
|
+
from rag_wright.spans.segment import segment_clause
|
|
612
|
+
|
|
613
|
+
seg = segment_clause
|
|
614
|
+
# segment (CPU/regex, sync) -> per-chunk operative spans, dropping empty chunks
|
|
615
|
+
per_chunk = [(ch, ops) for ch in chunks
|
|
616
|
+
if (ops := [op for op in seg(ch.chunk_id, ch.text) if op.text.strip()])]
|
|
617
|
+
if not per_chunk:
|
|
618
|
+
return []
|
|
619
|
+
n = max_concurrency if max_concurrency is not None else int(os.environ.get("CLASSIFY_CONCURRENCY", "8"))
|
|
620
|
+
sem = asyncio.Semaphore(n) # ONE shared bound on total in-flight classify calls (leaf-acquired -> no deadlock)
|
|
621
|
+
scores_by_chunk = await asyncio.gather(
|
|
622
|
+
*(classify_fn.aclassify_spans(ch.text, [op.text for op in ops], sem=sem) for ch, ops in per_chunk))
|
|
623
|
+
out: list = []
|
|
624
|
+
for (ch, ops), scores_per_span in zip(per_chunk, scores_by_chunk): # gather preserves order -> stable output
|
|
625
|
+
for op, scores in zip(ops, scores_per_span):
|
|
626
|
+
out.append((op, primary_function(scores) or NO_FUNCTION, ch.doc_start, scores))
|
|
627
|
+
return out
|
|
628
|
+
|
|
629
|
+
|
|
630
|
+
def clause_extraction_jobs(segments: list, boundary_starts: list[bool] | None = None) -> list:
|
|
631
|
+
"""Group ordered `segments` into PROVISIONS and emit one clause-extraction job per provision, as
|
|
632
|
+
`[(index, anchor_op, function, scores, text)]`: `text` is the provision's merged span text (what the extractor
|
|
633
|
+
reads), `anchor_op` is its first span (the citation anchor + provenance), `index` is the provision ordinal.
|
|
634
|
+
|
|
635
|
+
Issue 0038: a `Clause` is a PROVISION, not a sentence. Extracting per span made 98% of spans clauses (a clause
|
|
636
|
+
per sentence) once issue 0036 removed the function gate -- the gate had been doing accidental provision
|
|
637
|
+
detection. The provision unit is the numbered section (`spans.segment.starts_new_provision`); retrieval stays
|
|
638
|
+
per span (`index_fn` is unchanged). A provision boundary is a CHUNK change OR a heading span, so granularity
|
|
639
|
+
self-adjusts: numbered sections -> provision-level; a heading-less document -> chunk-level (never per sentence,
|
|
640
|
+
never one clause per document).
|
|
641
|
+
|
|
642
|
+
Within a provision, `is_extractable_span` still drops furniture spans (page numbers, signature/notice labels)
|
|
643
|
+
from the merged text; a provision that is ALL furniture yields no clause. The FUNCTION stays a soft tag
|
|
644
|
+
(ADR-0082, issue 0036): the provision's function is the first non-NONE among its spans, else NO_FUNCTION -- an
|
|
645
|
+
untagged provision is still extracted (function-independent extraction).
|
|
646
|
+
"""
|
|
647
|
+
from rag_wright.contracts.function import NO_FUNCTION
|
|
648
|
+
from rag_wright.spans.segment import is_extractable_span, starts_new_provision
|
|
649
|
+
|
|
650
|
+
# 1) group consecutive segments into provisions (boundary = chunk change OR a provision-heading span)
|
|
651
|
+
# `boundary_starts` (when provided) is the per-span "starts a new provision?" decision from
|
|
652
|
+
# `spans.boundary.adecide_provision_starts` (deterministic + the Jev residue fallback); None -> deterministic only.
|
|
653
|
+
groups: list[list] = []
|
|
654
|
+
for i, seg in enumerate(segments):
|
|
655
|
+
op = seg[0]
|
|
656
|
+
starts = boundary_starts[i] if boundary_starts is not None else starts_new_provision(op.text)
|
|
657
|
+
if groups and op.parent_chunk_id == groups[-1][-1][0].parent_chunk_id and not starts:
|
|
658
|
+
groups[-1].append(seg)
|
|
659
|
+
else:
|
|
660
|
+
groups.append([seg])
|
|
661
|
+
|
|
662
|
+
# 2) one job per provision, over its EXTRACTABLE spans only (furniture dropped; all-furniture -> no clause)
|
|
663
|
+
jobs: list = []
|
|
664
|
+
index = 0
|
|
665
|
+
for group in groups:
|
|
666
|
+
members = [seg for seg in group if is_extractable_span(seg[0].text)]
|
|
667
|
+
if not members:
|
|
668
|
+
continue
|
|
669
|
+
anchor_op = members[0][0]
|
|
670
|
+
text = "\n".join(seg[0].text.strip() for seg in members)
|
|
671
|
+
function = next((fn for (_op, fn, _cds, _sc) in members if fn and fn != NO_FUNCTION), NO_FUNCTION)
|
|
672
|
+
scores = members[0][3]
|
|
673
|
+
jobs.append((index, anchor_op, function, scores, text))
|
|
674
|
+
index += 1
|
|
675
|
+
return jobs
|
|
676
|
+
|
|
677
|
+
|
|
678
|
+
async def _aextract_clause_with_retry(
|
|
679
|
+
extractor: Any, *, chunk_id: Any, function: str, text: str, span_id: str, attempts: int,
|
|
680
|
+
functions: tuple[str, ...] = ()
|
|
681
|
+
) -> tuple[Any, str]:
|
|
682
|
+
"""Extract one clause with bounded retries. PARTIAL-CAUSE-1: docling-graph's `ExtractionFailed` is raised on
|
|
683
|
+
ANY logged docling error -- not only a deterministic "No valid JSON", but also TRANSIENT blips (an LLM empty
|
|
684
|
+
response, a gleaning failure, a rate-limit, a timeout). So EVERY failure is retried (the transient is what the
|
|
685
|
+
retry recovers); an earlier EXTRACT-GUARD-1 attempt to skip retrying `ExtractionFailed` turned recoverable
|
|
686
|
+
blips into lost clauses. The furniture that used to hard-fail deterministically is filtered UPSTREAM by
|
|
687
|
+
`is_extractable_span`, so this loop no longer retry-storms on non-clauses. Returns `(record, "")` on success or
|
|
688
|
+
`(None, reason)` on persistent failure."""
|
|
689
|
+
reason = ""
|
|
690
|
+
for _attempt in range(attempts):
|
|
691
|
+
try:
|
|
692
|
+
record = await extractor.aextract(chunk_id=chunk_id, function=function, text=text, span_id=span_id,
|
|
693
|
+
functions=functions)
|
|
694
|
+
return record, ""
|
|
695
|
+
except Exception as exc: # noqa: BLE001 - retry any failure (ExtractionFailed captures transients too)
|
|
696
|
+
reason = str(exc)
|
|
697
|
+
return None, reason
|
|
698
|
+
|
|
699
|
+
|
|
700
|
+
def _resolve_ingest_knobs(*, classify_concurrency: Any, clause_concurrency: Any, affiliations: Any,
|
|
701
|
+
function_classifier: Any) -> tuple:
|
|
702
|
+
"""EP-API-4a: resolve the four ingest knobs, `None` -> the engine default (env fallback, so a non-API caller is
|
|
703
|
+
unaffected), else the explicit config override. Returns `(classify_concurrency, clause_concurrency, affiliations,
|
|
704
|
+
function_classifier_kind)`. `classify_concurrency` is passed through as-is (the segment leaf falls back to env
|
|
705
|
+
when it is None), so the whole chain keeps one env default per knob."""
|
|
706
|
+
import os
|
|
707
|
+
|
|
708
|
+
return (
|
|
709
|
+
classify_concurrency,
|
|
710
|
+
clause_concurrency if clause_concurrency is not None else int(os.environ.get("CLAUSE_CONCURRENCY", "8")),
|
|
711
|
+
affiliations if affiliations is not None else (os.getenv("RAG_INGEST_AFFILIATIONS", "1") != "0"),
|
|
712
|
+
(function_classifier or os.getenv("RAG_FUNCTION_CLASSIFIER", "setfit")).lower(),
|
|
713
|
+
)
|
|
714
|
+
|
|
715
|
+
|
|
716
|
+
def aproduction_document_ingest(
|
|
717
|
+
store: Any, *, cache_dir: Any, registry: Any, embedder: Any = None, party_seed_path: Any = None,
|
|
718
|
+
classify_fn: Any = None, extract_model: Any = None, list_model: Any = None, samples: Any = None,
|
|
719
|
+
graph_extract_model: Any = None, judge_model: Any = None, chunk_model: Any = None,
|
|
720
|
+
classify_concurrency: Any = None, clause_concurrency: Any = None, affiliations: Any = None,
|
|
721
|
+
function_classifier: Any = None, embedding_profile: str = "bge-m3"):
|
|
722
|
+
"""ASYNC-B2e (ADR-0057): the async twin of `production_document_ingest`. Wires the ASYNC stage seams (achunk,
|
|
723
|
+
aclassify_spans, clause_extractor.aextract, aper_contract_graph_extraction) so the ingest model calls run on
|
|
724
|
+
the async seam with the true wall-clock deadline; CPU/store work (embed, resolve, DB writes) runs off the loop
|
|
725
|
+
via `asyncio.to_thread`. Clause extraction is bounded-concurrent via `asyncio.gather` + a `Semaphore`. Returns
|
|
726
|
+
an ASYNC per-document graph -- drive it with `arun_corpus_ingestion`.
|
|
727
|
+
|
|
728
|
+
Model configuration (mirrors the query/compliance entrypoints -- a caller no longer has to reach for env
|
|
729
|
+
vars to change the ingest models):
|
|
730
|
+
- `extract_model`: the PRIMARY clause-property extraction model -- an `ExtractionModel` OR a bare model-id
|
|
731
|
+
string (wrapped via `default_extraction_model`). `None` keeps the backend default (granite, per
|
|
732
|
+
`RAG_SERVING`). This is the model that produces the typed clause properties.
|
|
733
|
+
- `list_model`: the SECOND model for the cross-model UNION on the LIST-bearing groups only (carve_out /
|
|
734
|
+
covered_subject / damage_type). granite and gemma under-enumerate DIFFERENT list items, so their union
|
|
735
|
+
is more complete than either alone; the second model is cost-scoped to list groups. A bare model-id
|
|
736
|
+
string, `"off"` to disable, or `None` for the default (gemma, `RAG_INGEST_LIST_MODEL`). If you override
|
|
737
|
+
`extract_model` (e.g. to qwen), set `list_model` deliberately -- the union's value depends on the two
|
|
738
|
+
models being complementary.
|
|
739
|
+
- `samples`: same-model multi-sample count for the list union (`None` -> env `RAG_INGEST_CLAUSE_SAMPLES`,
|
|
740
|
+
default 1).
|
|
741
|
+
- `graph_extract_model`: the model for BOTH party AND affiliation extraction (the GP-1B graph-extract
|
|
742
|
+
surface -- they share one model). A bare model-id string or an `ExtractionModel` (unwrapped to its id);
|
|
743
|
+
`None` -> the default (granite, `RAG_GRAPH_EXTRACT_MODEL`).
|
|
744
|
+
- `judge_model`: the ingest semantic-judge model (ADR-0040 Layer-3 gate) -- a model-id string or an
|
|
745
|
+
`ExtractionModel`; `None` -> `model_for(STRUCTURED_REASONING)`.
|
|
746
|
+
- `chunk_model`: the chunker's boundary-refinement model -- structural boundaries are deterministic (zero
|
|
747
|
+
calls); ONLY an over-cap section triggers a bounded per-section tag-parse call, and this is the model it
|
|
748
|
+
uses. A model-id string or an `ExtractionModel`; `None` -> `model_for(GENERAL)`.
|
|
749
|
+
EP-API-4a ingest knobs (each `None` -> the env/default, so existing callers are unaffected):
|
|
750
|
+
`classify_concurrency` (function-classify parallelism), `clause_concurrency` (clause-extraction parallelism),
|
|
751
|
+
`affiliations` (run affiliation extraction), `function_classifier` ("setfit" | "llm"). The engine API passes
|
|
752
|
+
these from `EngineConfig.options.ingest`.
|
|
753
|
+
Env vars remain the fallback for every knob, so existing callers are unaffected."""
|
|
754
|
+
import asyncio
|
|
755
|
+
import hashlib
|
|
756
|
+
import json
|
|
757
|
+
from pathlib import Path
|
|
758
|
+
|
|
759
|
+
from rag_wright.capabilities.disambiguation import disambiguate
|
|
760
|
+
from rag_wright.capabilities.embedding_profiles import build_ingest_embedder
|
|
761
|
+
from rag_wright.capabilities.entity_resolution import resolve_entities
|
|
762
|
+
from rag_wright.capabilities.graph_extraction import aproduction_extract_fn
|
|
763
|
+
from rag_wright.capabilities.graph_storage import to_graph
|
|
764
|
+
from rag_wright.capabilities.rlm_chunking import StructuralModelFallbackDiscoverer, achunk
|
|
765
|
+
from rag_wright.contracts.contract_meta import ContractRecord
|
|
766
|
+
from rag_wright.contracts.identifiers import ChunkId
|
|
767
|
+
from rag_wright.contracts.property import ClausePropertyRecord
|
|
768
|
+
from rag_wright.models.profiles import ModelRole, model_for
|
|
769
|
+
from rag_wright.ontology.clause_template import Clause
|
|
770
|
+
from rag_wright.capabilities.dg_extraction import default_extraction_model
|
|
771
|
+
from rag_wright.spans.clause_kg_extractor import classifier_property_extractor
|
|
772
|
+
from rag_wright.spans.model_capabilities import (
|
|
773
|
+
CapabilityFunctionClassifier,
|
|
774
|
+
capability_property_classifier_fn,
|
|
775
|
+
)
|
|
776
|
+
from rag_wright.spans.segment import to_span_record
|
|
777
|
+
|
|
778
|
+
parse_dir = Path(cache_dir) / "parsed"
|
|
779
|
+
chunk_dir = Path(cache_dir) / "chunks"
|
|
780
|
+
clause_cache_dir = Path(cache_dir) / "clause_extract"
|
|
781
|
+
party_dir = Path(cache_dir) / "graph_parties"
|
|
782
|
+
affil_dir = Path(cache_dir) / "graph_affiliations" # issue 0027: separate from the party cache
|
|
783
|
+
for directory in (parse_dir, chunk_dir, clause_cache_dir, party_dir, affil_dir):
|
|
784
|
+
directory.mkdir(parents=True, exist_ok=True)
|
|
785
|
+
if party_seed_path is not None:
|
|
786
|
+
seed_party_cache(party_dir, party_seed_path)
|
|
787
|
+
# ADR-0058 (issue 0004): structure-first default -- deterministic boundaries from docling labels where present
|
|
788
|
+
# (zero model calls), bounded per-section TAG-PARSE fallback for over-cap sections (never the server-side
|
|
789
|
+
# guided-decoding whole-doc call that ran away past the 180s deadline). NOTE: this ingest currently flattens
|
|
790
|
+
# to text (`_parsed_from_text`), so labels are absent here and only the tag-parse fallback fires; preserving
|
|
791
|
+
# docling structure through ingest (a follow-up) unlocks the full zero-model structural win.
|
|
792
|
+
# the chunker's boundary-refinement model: structural boundaries are deterministic (zero calls); only an
|
|
793
|
+
# OVER-CAP section triggers a bounded per-section tag-parse call, and this is the model it uses (issue 0033
|
|
794
|
+
# follow-up). None -> default (GENERAL role). A bare id or an ExtractionModel (unwrapped to its id).
|
|
795
|
+
chunk_model_id = getattr(chunk_model, "model", chunk_model)
|
|
796
|
+
discoverer = StructuralModelFallbackDiscoverer(chunk_model_id)
|
|
797
|
+
summarizer = _NoSummary()
|
|
798
|
+
from rag_wright.spans.semantic_judge import build_asemantic_judge_fn
|
|
799
|
+
# caller-configurable ingest models (else backend/env defaults). A bare id -> an ExtractionModel; for the
|
|
800
|
+
# graph/judge surfaces (which take a model-id string) an ExtractionModel is unwrapped to its `.model` id.
|
|
801
|
+
clause_model = extract_model
|
|
802
|
+
if isinstance(extract_model, str):
|
|
803
|
+
clause_model = default_extraction_model("clause-extract", extract_model)
|
|
804
|
+
graph_extract_id = getattr(graph_extract_model, "model", graph_extract_model) # None or a bare model-id
|
|
805
|
+
judge_id = getattr(judge_model, "model", judge_model) or model_for(ModelRole.STRUCTURED_REASONING)
|
|
806
|
+
# CLS-D (ADR-0115): Step-3a property extraction is the classifier-first path -- the 29-dim best-of-both fleet,
|
|
807
|
+
# ONE residual LLM call for the 7 numeric/open dims (`clause_model`). EP-RT-7: the classifier LANE is dispatched
|
|
808
|
+
# through the `clause_property_classification` CAPABILITY (the single production path), never a second hand-built
|
|
809
|
+
# fleet here; the residual LLM call + the ADR-0028/0040/Layer-3 judge gates compose around it (ClassifierPropertyExtractor).
|
|
810
|
+
clause_extractor = classifier_property_extractor(
|
|
811
|
+
classifier_fn=capability_property_classifier_fn(),
|
|
812
|
+
model_id=getattr(clause_model, "model", clause_model), # the residual 7-numeric structured call
|
|
813
|
+
asemantic_judge_fn=build_asemantic_judge_fn(judge_id))
|
|
814
|
+
# party AND affiliation extraction share the graph-extract model (GP-1B); one arg drives both
|
|
815
|
+
aextract_parties_fn = (aproduction_extract_fn(model_id=graph_extract_id) if graph_extract_id
|
|
816
|
+
else aproduction_extract_fn())
|
|
817
|
+
# EP-API-4a: resolve the ingest knobs ONCE (config override else env/default), then use the resolved values.
|
|
818
|
+
_classify_concurrency, clause_concurrency, _affiliations_on, _clf_kind = _resolve_ingest_knobs(
|
|
819
|
+
classify_concurrency=classify_concurrency, clause_concurrency=clause_concurrency,
|
|
820
|
+
affiliations=affiliations, function_classifier=function_classifier)
|
|
821
|
+
if classify_fn is None:
|
|
822
|
+
# T55/SETFIT-SEG-1: the clause-function classifier is a SOFT tag (ADR-0047), so its implementation swaps
|
|
823
|
+
# behind this seam with NO contract/API change. DEFAULT is now the in-process trained SetFit ensemble
|
|
824
|
+
# soft-tagger (ms/span, no LLM call -- the ingestion-latency lever). `function_classifier="llm"` (or env
|
|
825
|
+
# RAG_FUNCTION_CLASSIFIER=llm) reverts to the LLM tag-classifier; a passed-in `classify_fn` overrides all.
|
|
826
|
+
if _clf_kind == "setfit":
|
|
827
|
+
# EP-RT-7: the default clause-function classifier dispatches through the `clause_function_classification`
|
|
828
|
+
# CAPABILITY (the single production path) -- not a second hand-built SetFit instance. (RAG_FUNCTION_CLASSIFIER=llm
|
|
829
|
+
# or an injected classify_fn are explicit non-capability overrides.)
|
|
830
|
+
classify_fn = CapabilityFunctionClassifier()
|
|
831
|
+
else:
|
|
832
|
+
from rag_wright.spans.clause_function_classifier import production_batch_clause_classifier
|
|
833
|
+
|
|
834
|
+
classify_fn = production_batch_clause_classifier(model_for(ModelRole.FUNCTION_CLASSIFY))
|
|
835
|
+
embedder = embedder if embedder is not None else build_ingest_embedder(embedding_profile)
|
|
836
|
+
template_version = hashlib.sha256(
|
|
837
|
+
json.dumps(Clause.model_json_schema(), sort_keys=True).encode("utf-8")).hexdigest()[:12]
|
|
838
|
+
_CLAUSE_EXTRACT_ATTEMPTS = 3 # clause_concurrency resolved above (EP-API-4a)
|
|
839
|
+
|
|
840
|
+
async def _aparty_names(text: str) -> list:
|
|
841
|
+
parties = await aextract_parties_fn(text)
|
|
842
|
+
return [p.name for p in parties.parties] if parties is not None else []
|
|
843
|
+
|
|
844
|
+
# issue 0027: corporate-affiliation extraction (AFFILIATE_OF). Default ON -- the lexical pre-filter keeps it a
|
|
845
|
+
# no-op for contracts that state no affiliation; config `affiliations=False` (or RAG_INGEST_AFFILIATIONS=0)
|
|
846
|
+
# disables it entirely. `_affiliations_on` resolved above (EP-API-4a).
|
|
847
|
+
async def _aaffiliations(text: str) -> list:
|
|
848
|
+
from rag_wright.capabilities.graph_extraction import aextract_affiliations
|
|
849
|
+
|
|
850
|
+
if graph_extract_id: # same graph-extract model as party extraction (issue 0033 follow-up)
|
|
851
|
+
return await aextract_affiliations(text, model_id=graph_extract_id)
|
|
852
|
+
return await aextract_affiliations(text)
|
|
853
|
+
|
|
854
|
+
async def chunk_fn(doc: SourceDocument) -> list:
|
|
855
|
+
parsed = _parsed_for(doc, parse_dir) # CHUNK-7: real docling parse if provided, else a text-only parse
|
|
856
|
+
manifest = await achunk(parsed, summarizer=summarizer, cache_dir=chunk_dir, discoverer=discoverer)
|
|
857
|
+
return list(manifest.chunks)
|
|
858
|
+
|
|
859
|
+
async def segment_fn(doc: SourceDocument, chunks: list) -> list:
|
|
860
|
+
segments = await _asegment_and_classify(chunks, classify_fn, max_concurrency=_classify_concurrency)
|
|
861
|
+
# issue 0032 (CU-B5): attach source-page provenance to each span. Build a page<->char-offset map over the
|
|
862
|
+
# canonical text (from the parsed doc's per-item prov pages) and look up each span's [doc_start, doc_end).
|
|
863
|
+
# Deterministic, no model call; degrades to no pages when the parse carried no provenance (text-only leg).
|
|
864
|
+
return _attach_page_provenance(doc, chunks, segments, parse_dir)
|
|
865
|
+
|
|
866
|
+
async def clauses_fn(doc: SourceDocument, segments: list) -> dict:
|
|
867
|
+
# issue 0038: a Clause is a PROVISION -- clause_extraction_jobs groups spans into provisions (numbered
|
|
868
|
+
# section, else chunk) and yields one job per provision (merged text + anchor span). Retrieval stays per
|
|
869
|
+
# span (index_fn unchanged). The function is a soft tag (ADR-0082), never a gate (issue 0036).
|
|
870
|
+
# Boundaries: deterministic-first, with a Jev decision-model fallback for the UNCERTAIN residue only
|
|
871
|
+
# (spans.boundary) -- flexible on new heading styles, degrades to deterministic with no decision model.
|
|
872
|
+
from rag_wright.spans.boundary import adecide_provision_starts, jev_boundary_decider
|
|
873
|
+
|
|
874
|
+
boundary_starts = await adecide_provision_starts(
|
|
875
|
+
[seg[0].text for seg in segments], decider=jev_boundary_decider())
|
|
876
|
+
jobs = clause_extraction_jobs(segments, boundary_starts=boundary_starts)
|
|
877
|
+
if not jobs:
|
|
878
|
+
return {"clause_records": [], "clause_failures": []}
|
|
879
|
+
failures: list[dict] = []
|
|
880
|
+
sem = asyncio.Semaphore(clause_concurrency)
|
|
881
|
+
|
|
882
|
+
from rag_wright.contracts.function import NO_FUNCTION as _NO_FUNCTION
|
|
883
|
+
|
|
884
|
+
async def _extract(job: Any) -> Any:
|
|
885
|
+
index, anchor_op, function, scores, text = job # text = merged provision; anchor_op = citation anchor
|
|
886
|
+
# CLS-D soft-scoping: the anchor span's TOP-3 real function soft-tags scope the classifier lane (union
|
|
887
|
+
# of their dims) -- tolerant of the ~0.5 function accuracy, and kills the over-emission a classifier
|
|
888
|
+
# (which cannot abstain) causes when run unscoped. An untagged span -> () -> no scoping (every dim runs).
|
|
889
|
+
functions = tuple(dict.fromkeys(
|
|
890
|
+
s.function for s in (scores or []) if s.function and s.function != _NO_FUNCTION))[:3]
|
|
891
|
+
clause_cid = ChunkId.of(doc.source_doc_id, index, text)
|
|
892
|
+
cache_file = clause_cache_dir / (hashlib.sha256(
|
|
893
|
+
f"{clause_cid.value}|{function}|{template_version}".encode("utf-8")).hexdigest()[:32] + ".json")
|
|
894
|
+
if cache_file.exists(): # a prior SUCCESSFUL extraction -> reuse it, no granite re-call
|
|
895
|
+
record = ClausePropertyRecord.model_validate_json(cache_file.read_text(encoding="utf-8"))
|
|
896
|
+
else:
|
|
897
|
+
record, reason = await _aextract_clause_with_retry(
|
|
898
|
+
clause_extractor, chunk_id=clause_cid, function=function, text=text,
|
|
899
|
+
span_id=anchor_op.span_id, attempts=_CLAUSE_EXTRACT_ATTEMPTS, functions=functions)
|
|
900
|
+
if record is None: # persistent failure -> record it (PARTIAL), do NOT cache, do NOT silently drop
|
|
901
|
+
failures.append({"span_id": anchor_op.span_id, "function": function, "reason": reason[:200]})
|
|
902
|
+
return None
|
|
903
|
+
cache_file.write_text(record.model_dump_json(), encoding="utf-8")
|
|
904
|
+
return record.model_copy(update={"functions": scores})
|
|
905
|
+
|
|
906
|
+
async def _bounded(job: Any) -> Any:
|
|
907
|
+
async with sem: # backpressure (network-bound granite)
|
|
908
|
+
return await _extract(job)
|
|
909
|
+
|
|
910
|
+
results = [r for r in await asyncio.gather(*(_bounded(j) for j in jobs)) if r is not None]
|
|
911
|
+
return {"clause_records": results, "clause_failures": failures}
|
|
912
|
+
|
|
913
|
+
async def index_fn(doc: SourceDocument, segments: list) -> dict:
|
|
914
|
+
if not segments:
|
|
915
|
+
return {"span_count": 0, "span_failures": []}
|
|
916
|
+
dense_vecs, sparse_vecs = await asyncio.to_thread(
|
|
917
|
+
embedder.encode_batch, [op.text.strip() for op, _, _, _ in segments])
|
|
918
|
+
|
|
919
|
+
def _write_all() -> dict:
|
|
920
|
+
count = 0
|
|
921
|
+
failures: list[dict] = []
|
|
922
|
+
for (op, function, chunk_doc_start, scores), dense, sparse in zip(segments, dense_vecs, sparse_vecs):
|
|
923
|
+
try:
|
|
924
|
+
store.upsert_span(to_span_record(
|
|
925
|
+
op, contract_id=doc.source_doc_id, chunk_doc_start=chunk_doc_start,
|
|
926
|
+
dense_vector=list(dense), sparse_vector=sparse, function=function,
|
|
927
|
+
functions=[s.function for s in scores])) # T55: top-k soft tags (primary-first)
|
|
928
|
+
count += 1
|
|
929
|
+
except Exception as exc: # noqa: BLE001 - a per-span write must not sink the KG, but is NOT swallowed
|
|
930
|
+
failures.append({"span_id": op.span_id, "reason": repr(exc)}) # 0006-C: surfaced -> PARTIAL
|
|
931
|
+
return {"span_count": count, "span_failures": failures}
|
|
932
|
+
|
|
933
|
+
return await asyncio.to_thread(_write_all)
|
|
934
|
+
|
|
935
|
+
async def graph_fn(doc: SourceDocument, chunks: list) -> list: # noqa: ARG001 - GP-1B is per-CONTRACT
|
|
936
|
+
return await aper_contract_graph_extraction(
|
|
937
|
+
doc, party_dir=party_dir, anames_fn=_aparty_names,
|
|
938
|
+
affil_dir=(affil_dir if _affiliations_on else None),
|
|
939
|
+
aaffiliations_fn=(_aaffiliations if _affiliations_on else None))
|
|
940
|
+
|
|
941
|
+
async def resolve_fn(extraction_results: list) -> Any:
|
|
942
|
+
return await asyncio.to_thread(
|
|
943
|
+
lambda: to_graph(resolve_entities(
|
|
944
|
+
disambiguate(extraction_results), extraction_results, resolver=registry))) # registry IS an EntityResolver (DD-3)
|
|
945
|
+
|
|
946
|
+
async def write_fn(doc: SourceDocument, clause_records: list, resolution: Any) -> dict:
|
|
947
|
+
from rag_wright.capabilities.contract_kg_store import ContractKGStore # DD-1b: clause KG + contract meta
|
|
948
|
+
|
|
949
|
+
ckg = ContractKGStore(store)
|
|
950
|
+
|
|
951
|
+
def _write() -> dict:
|
|
952
|
+
for record in clause_records:
|
|
953
|
+
ckg.write_clause_kg(record)
|
|
954
|
+
nodes, edges = resolution
|
|
955
|
+
store.write_graph(nodes, edges)
|
|
956
|
+
ckg.upsert_contract(ContractRecord(
|
|
957
|
+
contract_id=doc.source_doc_id, name=doc.metadata.get("raw_title", ""),
|
|
958
|
+
source_doc_id=doc.source_doc_id,
|
|
959
|
+
content_hash=hashlib.sha256(doc.text.encode("utf-8")).hexdigest()))
|
|
960
|
+
return {"clauses": len(clause_records), "entities": len(nodes), "edges": len(edges)}
|
|
961
|
+
|
|
962
|
+
return await asyncio.to_thread(_write)
|
|
963
|
+
|
|
964
|
+
return abuild_document_ingest(
|
|
965
|
+
chunk_fn, segment_fn, clauses_fn, index_fn, graph_fn, resolve_fn, write_fn)
|
|
966
|
+
|
|
967
|
+
|
|
968
|
+
# (issue 0028 / ADR-0091: `corpus_party_link_fn` -- the KG-7 PartyTo link provider -- was retired with the
|
|
969
|
+
# PartyTo edge. `arun_corpus_ingestion(link_fn=...)` keeps its no-op default; there is no PartyTo provider.)
|
|
970
|
+
|
|
971
|
+
|
|
972
|
+
def register_contract_ingestion_pipeline(registry) -> None:
|
|
973
|
+
"""LG-3d: register `contract_ingestion_pipeline` (composite subgraph; source docs -> populated contract KG)."""
|
|
974
|
+
registry.register(
|
|
975
|
+
"contract_ingestion_pipeline",
|
|
976
|
+
contract=IngestionReport,
|
|
977
|
+
kind="subgraph",
|
|
978
|
+
display_name="Contract ingestion pipeline (corpus -> populated, connected KG)",
|
|
979
|
+
)
|
|
980
|
+
|
|
981
|
+
|
|
982
|
+
async def ainvoke(resources, inputs: dict):
|
|
983
|
+
"""EP-CORE-2 (ADR-0118): the capability invoke factory (impl_ref target) -- ingest ONE document through the
|
|
984
|
+
async per-document graph over the opaque handle. `inputs`: document (an api.source_document / parse_document
|
|
985
|
+
SourceDocument) + cache_dir. Ingest knobs + embedder come from EngineConfig.options/embeddings (EP-API-4a/4b);
|
|
986
|
+
the entity resolver is the generic closed-world default (DD-3)."""
|
|
987
|
+
from rag_wright.capabilities.dg_extraction import default_extraction_model
|
|
988
|
+
from rag_wright.models.profiles import ModelRole
|
|
989
|
+
from rag_wright.ontology.registry import EntityRegistry
|
|
990
|
+
|
|
991
|
+
opts = resources._config.options.ingest
|
|
992
|
+
graph = aproduction_document_ingest(
|
|
993
|
+
resources._store, cache_dir=inputs["cache_dir"], registry=EntityRegistry(),
|
|
994
|
+
extract_model=default_extraction_model(model=resources.model_id(ModelRole.STRUCTURED_REASONING)),
|
|
995
|
+
embedding_profile=resources._config.embeddings.get("text", "bge-m3"),
|
|
996
|
+
list_model=opts.list_model, samples=opts.clause_samples,
|
|
997
|
+
classify_concurrency=opts.classify_concurrency, clause_concurrency=opts.clause_concurrency,
|
|
998
|
+
affiliations=opts.affiliations, function_classifier=opts.function_classifier)
|
|
999
|
+
return await graph.ainvoke({"document": inputs["document"]})
|