rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,299 @@
|
|
|
1
|
+
"""DOCPARSE-1 (ADR-0049 generic-customer lens): the generic raw-document parser -- the SHARED front-end that lets a
|
|
2
|
+
customer's own PDF / DOCX / HTML flow into BOTH ingestion sides.
|
|
3
|
+
|
|
4
|
+
The docling parse capability already exists (`capabilities/parsing.py`: `DocumentConverter` -> `DoclingDocument`);
|
|
5
|
+
this module adds (a) a BYTES entry point (customer docs arrive as bytes, e.g. from GCS, not a local path) and the
|
|
6
|
+
two PROJECTIONS the two sides need from a parsed document:
|
|
7
|
+
- `document_to_text` -> the contract side (`GcsCorpusAdapter.parse_bytes` seam): one text blob per document.
|
|
8
|
+
- `document_to_sections` -> the compliance side (`RegulationAdapter`'s `[{section, heading, text}]` shape): the
|
|
9
|
+
document split into sections at its headings, so a policy PDF ingests the same way an eCFR `sections.json` does.
|
|
10
|
+
|
|
11
|
+
The DoclingDocument API is grounded (framework graph + installed `inspect`): `export_to_markdown()` for text;
|
|
12
|
+
`iterate_items() -> (item, level)` with heading labels SECTION_HEADER / TITLE / FIELD_HEADING (PAGE_HEADER is
|
|
13
|
+
page furniture, not a section boundary)."""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import re
|
|
17
|
+
import tempfile
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Any, Optional
|
|
21
|
+
|
|
22
|
+
from docling_core.types.doc.labels import DocItemLabel
|
|
23
|
+
|
|
24
|
+
# Labels that START a new section (a real heading), vs PAGE_HEADER which is running page furniture (ignored).
|
|
25
|
+
_HEADING_LABELS = frozenset({DocItemLabel.SECTION_HEADER, DocItemLabel.TITLE, DocItemLabel.FIELD_HEADING})
|
|
26
|
+
_TEXT_EXTS = frozenset({"txt", "md", "text"})
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
# 0009-WIRE2: a generous per-document OCR ceiling. A degraded multi-page doc escalated whole to the VLM is a
|
|
30
|
+
# batch op (~30s/page), so it needs more than the per-model-call deadline; per-page escalation would let us
|
|
31
|
+
# tighten this. Overridable via `deadline_s`.
|
|
32
|
+
_OCR_PARSE_DEADLINE_S = 600.0
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _default_document_parser() -> Any:
|
|
36
|
+
"""0009-WIRE2: the default parser for raw-document bytes is the TIERED OCR parser -- fast OCR, then a
|
|
37
|
+
scan-quality gate escalates only degraded pages to the VLM (default Gemma-4 via OpenRouter), and flags what
|
|
38
|
+
even the VLM cannot read. Graceful degrade when no VLM is configured. So BOTH ingestion pipelines and the MCP
|
|
39
|
+
document tool get degraded-scan handling through this one chokepoint."""
|
|
40
|
+
from rag_wright.capabilities.parsing import TieredOCRParser
|
|
41
|
+
|
|
42
|
+
return TieredOCRParser()
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def parse_document_bytes(name: str, data: bytes, *, parser: Any = None) -> Any:
|
|
46
|
+
"""Parse raw document BYTES into a `DoclingDocument`. `.txt`/`.md` bytes are wrapped directly; binary docs
|
|
47
|
+
(PDF/DOCX/HTML) go through docling. `parser` defaults to the tiered OCR parser (0009-WIRE2), injectable for
|
|
48
|
+
tests. `name` supplies the file extension docling needs to pick a backend."""
|
|
49
|
+
ext = name.rsplit(".", 1)[-1].lower() if "." in name else ""
|
|
50
|
+
parser = parser or _default_document_parser()
|
|
51
|
+
# docling reads a file path, not raw bytes -> write to a temp file preserving the extension for backend choice
|
|
52
|
+
suffix = f".{ext}" if ext else ".txt"
|
|
53
|
+
tmp = Path(tempfile.mkdtemp(prefix="docparse_")) / f"doc{suffix}"
|
|
54
|
+
tmp.write_bytes(data)
|
|
55
|
+
return parser.convert(tmp)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
async def aparse_document_bytes(name: str, data: bytes, *, parser: Any = None,
|
|
59
|
+
deadline_s: float = _OCR_PARSE_DEADLINE_S) -> Any:
|
|
60
|
+
"""ASYNC-bounded document parse (ADR-0057): run the sync `parse_document_bytes` (incl. the tiered VLM
|
|
61
|
+
escalation -- the slowest call in the pipeline) OFF the event loop via `to_thread`, under a wall-clock
|
|
62
|
+
`asyncio.timeout` so a hung/slow OCR never stalls the async ingestion. NOTE: `to_thread` cannot cancel the
|
|
63
|
+
worker thread, so the deadline unblocks the CALLER (raises TimeoutError); the docling parse thread finishes
|
|
64
|
+
in the background. True cancellation would require routing the vision call through the async model seam."""
|
|
65
|
+
import asyncio
|
|
66
|
+
|
|
67
|
+
async with asyncio.timeout(deadline_s):
|
|
68
|
+
return await asyncio.to_thread(parse_document_bytes, name, data, parser=parser)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def document_to_text(doc: Any) -> str:
|
|
72
|
+
"""A parsed document -> one text blob (docling markdown export) -- the contract side's `parse_bytes` output."""
|
|
73
|
+
return doc.export_to_markdown()
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
@dataclass(frozen=True)
|
|
77
|
+
class ContentItem:
|
|
78
|
+
"""issue 0014: one chunkable unit of a parsed document's READING-ORDER body. Duck-typed to a docling text
|
|
79
|
+
item (`.label`, `.level`, `.text`) so the chunker/segmenter consume it unchanged. The reading-order body is a
|
|
80
|
+
SUPERSET of `document.texts`: a docling TABLE lives in `document.tables` and a figure in `document.pictures`,
|
|
81
|
+
NEVER in `.texts`, so a `document.texts`-only chunker silently drops them (never chunked, never indexed, never
|
|
82
|
+
retrievable, and no failure recorded -- the exact 0014 loss). This projection is the single authority for
|
|
83
|
+
'the document's chunkable content, in reading order'.
|
|
84
|
+
|
|
85
|
+
issue 0032: each item also carries its parse-time PAGE provenance (`page`, 1-based, from `prov[0].page_no`)
|
|
86
|
+
and, when the parser produced one, a `bbox` (l, t, r, b on `page`). This is the provenance the CU-B5 page
|
|
87
|
+
map threads to a span's citation (a scanned-PDF click-through lands on the right page). `bbox` is best-effort
|
|
88
|
+
-- omitted where `prov` has none, and dropped when lines are merged into a paragraph (ambiguous then)."""
|
|
89
|
+
|
|
90
|
+
label: Any
|
|
91
|
+
level: Optional[int]
|
|
92
|
+
text: str
|
|
93
|
+
page: Optional[int] = None # issue 0032: 1-based source page (prov[0].page_no); None when prov is absent
|
|
94
|
+
bbox: Optional[tuple[float, float, float, float]] = None # (l, t, r, b) on `page`; only when prov has one
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _table_content_text(item: Any, doc: Any) -> str:
|
|
98
|
+
"""A TABLE item -> its atomic markdown (issue 0014), caption prefixed when docling captured one. The markdown
|
|
99
|
+
keeps the header row with the data rows, so the fee/payment schedule retrieves as a unit."""
|
|
100
|
+
caption = (item.caption_text(doc) or "").strip()
|
|
101
|
+
body = (item.export_to_markdown(doc) or "").strip()
|
|
102
|
+
return f"{caption}\n\n{body}".strip() if caption else body
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _picture_content_text(item: Any, doc: Any) -> str:
|
|
106
|
+
"""A PICTURE item -> its extractable text (issue 0014): the caption plus any description annotation
|
|
107
|
+
(a VLM/description the tiered OCR attached). Empty when the figure carries no text -- nothing to index."""
|
|
108
|
+
parts: list[str] = []
|
|
109
|
+
caption = (item.caption_text(doc) or "").strip()
|
|
110
|
+
if caption:
|
|
111
|
+
parts.append(caption)
|
|
112
|
+
for annotation in getattr(item, "annotations", None) or []:
|
|
113
|
+
text = (getattr(annotation, "text", "") or "").strip() # DescriptionAnnotation (figure description)
|
|
114
|
+
if text:
|
|
115
|
+
parts.append(text)
|
|
116
|
+
return "\n\n".join(parts)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
_ENDS_SENTENCE = re.compile(r"""[.:;?!]["')\]]*$""") # a line that completes a sentence (terminal punct + closers)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _ends_sentence(text: str) -> bool:
|
|
123
|
+
return bool(_ENDS_SENTENCE.search(text.rstrip()))
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _starts_new_sentence(text: str) -> bool:
|
|
127
|
+
"""A line that BEGINS a new provision: its first non-space char is a capital, a digit, or an opener
|
|
128
|
+
(`(`, `[`, quote, `§`, bullet). A lowercase start is a wrapped continuation ('and (ii) ...')."""
|
|
129
|
+
stripped = text.lstrip()
|
|
130
|
+
if not stripped:
|
|
131
|
+
return False
|
|
132
|
+
c = stripped[0]
|
|
133
|
+
return c.isupper() or c.isdigit() or c in "([{\"'§•-"
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _join_wrapped(prev: str, nxt: str) -> str:
|
|
137
|
+
"""Join a wrapped continuation to its paragraph: de-hyphenate a soft line-break (`Distribu-` + `tor` ->
|
|
138
|
+
`Distributor`), otherwise a single space."""
|
|
139
|
+
if prev.endswith("-") and len(prev) >= 2 and prev[-2].isalpha():
|
|
140
|
+
return prev[:-1] + nxt
|
|
141
|
+
return f"{prev} {nxt}"
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _merge_wrapped_lines(items: list[ContentItem]) -> list[ContentItem]:
|
|
145
|
+
"""DEFRAG-1: reconstruct paragraphs from docling's per-LINE text items. docling emits each PDF text line as its
|
|
146
|
+
own item; joined with `\\n\\n` and split by `segment_clause`, one clause shatters into per-line fragments that
|
|
147
|
+
then fail extraction (NEONSYSTEMS: 219 segments / 121 degenerate -> 87 / 9 after this pass). Consecutive
|
|
148
|
+
`TEXT` items are merged into one paragraph, breaking ONLY when the previous line ends a sentence AND the next
|
|
149
|
+
starts one (two-sided, so an abbreviation like 'Inc.' followed by a lowercase 'and' does not false-split, and a
|
|
150
|
+
clean paragraph-per-item document is left untouched). Any NON-text item (heading, table, figure, list item,
|
|
151
|
+
page furniture) is a hard boundary -- never merged across."""
|
|
152
|
+
out: list[ContentItem] = []
|
|
153
|
+
buf: str = ""
|
|
154
|
+
buf_level: Optional[int] = None
|
|
155
|
+
buf_page: Optional[int] = None # issue 0032: the merged paragraph keeps its FIRST line's page (bbox dropped)
|
|
156
|
+
for item in items:
|
|
157
|
+
if item.label != DocItemLabel.TEXT: # heading / table / picture / list-item / furniture -> hard boundary
|
|
158
|
+
if buf.strip():
|
|
159
|
+
out.append(ContentItem(label=DocItemLabel.TEXT, level=buf_level, text=buf, page=buf_page))
|
|
160
|
+
buf, buf_level, buf_page = "", None, None
|
|
161
|
+
out.append(item)
|
|
162
|
+
continue
|
|
163
|
+
line = (item.text or "").strip()
|
|
164
|
+
if not line:
|
|
165
|
+
continue
|
|
166
|
+
if not buf:
|
|
167
|
+
buf, buf_level, buf_page = line, item.level, item.page
|
|
168
|
+
elif _ends_sentence(buf) and _starts_new_sentence(line): # a real paragraph break
|
|
169
|
+
out.append(ContentItem(label=DocItemLabel.TEXT, level=buf_level, text=buf, page=buf_page))
|
|
170
|
+
buf, buf_level, buf_page = line, item.level, item.page
|
|
171
|
+
else: # a wrapped continuation of the same clause
|
|
172
|
+
buf = _join_wrapped(buf, line)
|
|
173
|
+
if buf.strip():
|
|
174
|
+
out.append(ContentItem(label=DocItemLabel.TEXT, level=buf_level, text=buf, page=buf_page))
|
|
175
|
+
return out
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _prov_page_bbox(node: Any) -> tuple[Optional[int], Optional[tuple[float, float, float, float]]]:
|
|
179
|
+
"""issue 0032: a docling item's page (1-based) and best-effort bbox from `prov[0]` (the same provenance
|
|
180
|
+
`scan_quality._page_texts` reads). Returns `(None, None)` when the item has no provenance (a test stub or a
|
|
181
|
+
born-item with none). The bbox is `(l, t, r, b)` when the parser produced one, else `None`."""
|
|
182
|
+
prov = getattr(node, "prov", None) or []
|
|
183
|
+
if not prov:
|
|
184
|
+
return None, None
|
|
185
|
+
page = getattr(prov[0], "page_no", None)
|
|
186
|
+
box = getattr(prov[0], "bbox", None)
|
|
187
|
+
bbox = None
|
|
188
|
+
if box is not None:
|
|
189
|
+
try:
|
|
190
|
+
bbox = (float(box.l), float(box.t), float(box.r), float(box.b))
|
|
191
|
+
except (AttributeError, TypeError, ValueError):
|
|
192
|
+
bbox = None
|
|
193
|
+
return (int(page) if page is not None else None), bbox
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _node_text(node: Any) -> str:
|
|
197
|
+
"""issue 0039: a docling ENUMERATED list item (a numbered contract provision) carries its number in `marker`
|
|
198
|
+
and STRIPS it from `.text` ('1.1.' + text -> text). The chunker/segmenter/provision-detector read the text,
|
|
199
|
+
so the section number vanishes and `starts_new_provision` never fires -> every provision falls back to the
|
|
200
|
+
chunk boundary (issue 0038 grouping degrades to chunk-level). Reconstruct the original by prepending the
|
|
201
|
+
marker (docling's own `orig`), so the number is present exactly where the detector looks. Only for an
|
|
202
|
+
`enumerated` node with a marker; a bullet/letter marker is prepended too (it restores the original and does
|
|
203
|
+
NOT trip the numeric section detector, so those list items still fold into their provision)."""
|
|
204
|
+
text = getattr(node, "text", "") or ""
|
|
205
|
+
marker = (getattr(node, "marker", "") or "").strip()
|
|
206
|
+
if marker and getattr(node, "enumerated", False) and not text.lstrip().startswith(marker):
|
|
207
|
+
return f"{marker} {text}".strip()
|
|
208
|
+
return text
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def content_items(doc: Any) -> list[ContentItem]:
|
|
212
|
+
"""A parsed document -> its READING-ORDER chunkable content items (issue 0014). Walks `iterate_items` over the
|
|
213
|
+
BODY and FURNITURE layers (so everything in `document.texts`, incl. page furniture, is covered), mapping each
|
|
214
|
+
item to a `ContentItem`: a TABLE -> its atomic markdown, a PICTURE -> caption + description text, any other
|
|
215
|
+
item -> its `.text`. A table/picture with no extractable text yields an empty-text item (the chunker strips it
|
|
216
|
+
exactly as it strips an empty text item today) -- present in reading order, never silently missing.
|
|
217
|
+
|
|
218
|
+
NO-SILENT-LOSS GUARANTEE (0014, the issue's Q2): any `document.texts` item the reading-order walk did not
|
|
219
|
+
visit is appended, so the projection is a strict SUPERSET of `.texts` -- a content item can never vanish
|
|
220
|
+
upstream of the failure accounting the way a table did before this fix."""
|
|
221
|
+
from docling_core.types.doc.document import ContentLayer
|
|
222
|
+
|
|
223
|
+
def _text_item(node: Any) -> ContentItem:
|
|
224
|
+
page, bbox = _prov_page_bbox(node)
|
|
225
|
+
return ContentItem(label=getattr(node, "label", None), level=getattr(node, "level", None),
|
|
226
|
+
text=_node_text(node), page=page, bbox=bbox) # issue 0039: keep the enumerated marker
|
|
227
|
+
|
|
228
|
+
if not hasattr(doc, "iterate_items"): # a plain `.texts`-bearing view (a `_SubDocument` slice / test stub):
|
|
229
|
+
return [_text_item(t) for t in getattr(doc, "texts", None) or []] # no reading-order body, no tables
|
|
230
|
+
|
|
231
|
+
layers = {ContentLayer.BODY, ContentLayer.FURNITURE}
|
|
232
|
+
items: list[ContentItem] = []
|
|
233
|
+
seen: set[int] = set()
|
|
234
|
+
for node, _level in doc.iterate_items(included_content_layers=layers):
|
|
235
|
+
seen.add(id(node))
|
|
236
|
+
label = getattr(node, "label", None)
|
|
237
|
+
if label == DocItemLabel.TABLE and hasattr(node, "export_to_markdown"):
|
|
238
|
+
text = _table_content_text(node, doc)
|
|
239
|
+
elif label == DocItemLabel.PICTURE:
|
|
240
|
+
text = _picture_content_text(node, doc)
|
|
241
|
+
else:
|
|
242
|
+
text = _node_text(node) # issue 0039: an enumerated list item keeps its section-number marker
|
|
243
|
+
page, bbox = _prov_page_bbox(node)
|
|
244
|
+
items.append(ContentItem(label=label, level=getattr(node, "level", None), text=text,
|
|
245
|
+
page=page, bbox=bbox))
|
|
246
|
+
items = _merge_wrapped_lines(items) # DEFRAG-1: rejoin per-line items into whole-clause paragraphs
|
|
247
|
+
for text_item in getattr(doc, "texts", None) or []: # coverage backstop: never drop a `.texts` item
|
|
248
|
+
if id(text_item) not in seen:
|
|
249
|
+
items.append(_text_item(text_item))
|
|
250
|
+
return items
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def _section_number(heading: str, index: int) -> str:
|
|
254
|
+
"""A short section id: the leading numeric token of the heading (e.g. '1' from '1. Confidentiality'), else the
|
|
255
|
+
1-based position -- so the compliance citation is stable and human-meaningful."""
|
|
256
|
+
token = heading.strip().split()[0].rstrip(".").rstrip(")") if heading.strip() else ""
|
|
257
|
+
return token if token and any(c.isdigit() for c in token) else str(index)
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def document_to_sections(doc: Any) -> list[dict]:
|
|
261
|
+
"""A parsed document -> `[{section, heading, text, pages, bbox}]` split at its headings -- the compliance side's
|
|
262
|
+
shape (`RegulationAdapter` ingests exactly this). Body before the first heading is kept as a leading section
|
|
263
|
+
(heading ""), so nothing is dropped. PAGE_HEADER and whitespace-only items are skipped.
|
|
264
|
+
|
|
265
|
+
issue 0043: each section carries `pages` (the source page(s) its items span, from the parse's per-item
|
|
266
|
+
provenance -- present even on a scan) and a best-effort `bbox`: a single-item section reports that item's box,
|
|
267
|
+
a multi-item section reports None (a section is not one rectangle, and a fabricated box is worse than none)."""
|
|
268
|
+
sections: list[dict] = []
|
|
269
|
+
heading = ""
|
|
270
|
+
body: list[str] = []
|
|
271
|
+
pages: set[int] = set()
|
|
272
|
+
boxes: list[Any] = [] # the (page, bbox) of each item, to pick a single-item section's box
|
|
273
|
+
|
|
274
|
+
def _flush() -> None:
|
|
275
|
+
text = "\n".join(body).strip()
|
|
276
|
+
if heading or text: # keep a section if it has a heading OR any body (never emit a fully empty one)
|
|
277
|
+
bbox = boxes[0] if len(boxes) == 1 else None # best-effort: only a single-item section has one box
|
|
278
|
+
sections.append({"section": _section_number(heading, len(sections) + 1), "heading": heading,
|
|
279
|
+
"text": text, "pages": sorted(pages), "bbox": bbox})
|
|
280
|
+
|
|
281
|
+
for item, _level in doc.iterate_items():
|
|
282
|
+
label = getattr(item, "label", None)
|
|
283
|
+
text = (getattr(item, "text", "") or "").strip()
|
|
284
|
+
page, box = _prov_page_bbox(item)
|
|
285
|
+
if label in _HEADING_LABELS:
|
|
286
|
+
_flush() # close the previous section
|
|
287
|
+
heading, body, pages, boxes = text, [], set(), []
|
|
288
|
+
if page is not None:
|
|
289
|
+
pages.add(page) # the heading's page belongs to its section
|
|
290
|
+
elif label == DocItemLabel.PAGE_HEADER:
|
|
291
|
+
continue # running page furniture -> neither a boundary nor body
|
|
292
|
+
elif text:
|
|
293
|
+
body.append(text)
|
|
294
|
+
if page is not None:
|
|
295
|
+
pages.add(page)
|
|
296
|
+
if box is not None:
|
|
297
|
+
boxes.append(box)
|
|
298
|
+
_flush() # the final section
|
|
299
|
+
return sections
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"""EDGAR entity acquisition logic (T7, docs/archive/plans/Corpus_Acquisition.md).
|
|
2
|
+
|
|
3
|
+
Pure logic; the throttled/cached fetch and the CLI live in `scripts/acquire_edgar.py`.
|
|
4
|
+
|
|
5
|
+
`normalize_cik` is the **single** canonical CIK->EntityId normalization point: a CIK is 10-digit
|
|
6
|
+
zero-padded, which is exactly the canonical `EntityId` (T1), so normalization is zero-pad-to-10 then
|
|
7
|
+
validate against the strict contract. T8's registry loader **reuses this exact function** so the two
|
|
8
|
+
cannot drift from the T1 contract.
|
|
9
|
+
|
|
10
|
+
Name->CIK proposals are mechanical and structurally **UNVERIFIED**: linking a contract party to a
|
|
11
|
+
CIK is itself the entity-resolution problem (FR-C.7), so a fuzzy/mechanical proposal is never ground
|
|
12
|
+
truth. Each proposal carries an explicit `status` (T7 only ever emits UNVERIFIED; VERIFIED is set by
|
|
13
|
+
human verification at T10), and the CLI writes proposals to a separate `proposed/` location. The
|
|
14
|
+
`MatchCoverage` records which parties resolved and which did not, so T10 verification starts from a
|
|
15
|
+
known map.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import re
|
|
21
|
+
from enum import Enum
|
|
22
|
+
|
|
23
|
+
from pydantic import BaseModel
|
|
24
|
+
|
|
25
|
+
from rag_wright.contracts.identifiers import EntityId
|
|
26
|
+
|
|
27
|
+
_CIK_PREFIX = re.compile(r"(?i)^cik[-:_ ]*")
|
|
28
|
+
_NON_ALNUM = re.compile(r"[^a-z0-9]+")
|
|
29
|
+
|
|
30
|
+
# Read this before treating the unresolved set as "not entities" (a note for the T10 verifier).
|
|
31
|
+
# Conservative normalized-conformed-name matching systematically leaves two classes unresolved, by
|
|
32
|
+
# design, not as a bug: (1) private companies and individuals, who are not in EDGAR at all (only
|
|
33
|
+
# public filers are), so a public-company-to-private-counterparty contract resolves one side only;
|
|
34
|
+
# and (2) name variants not close to the conformed name (subsidiaries filing under a parent, former
|
|
35
|
+
# names, DBAs), some of which the submissions former-names data can later help resolve. Both are
|
|
36
|
+
# exactly the cases human verification (T10) exists for. Consequence for the eval: graph entity
|
|
37
|
+
# coverage will be public-filer-centric (FR-C.7), which matters when reading multi-hop results.
|
|
38
|
+
UNRESOLVED_NOTE = (
|
|
39
|
+
"Unresolved does NOT mean 'not an entity'. Conservative conformed-name matching intentionally "
|
|
40
|
+
"misses private companies / individuals (absent from EDGAR) and name variants (subsidiaries, "
|
|
41
|
+
"former names, DBAs). These are the cases human verification (T10) exists for; graph entity "
|
|
42
|
+
"coverage is therefore public-filer-centric (FR-C.7)."
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def normalize_cik(raw: int | str) -> EntityId:
|
|
47
|
+
"""Normalize a raw EDGAR CIK (int, unpadded, or ``CIK``-prefixed) to a canonical `EntityId`.
|
|
48
|
+
|
|
49
|
+
This is the one place messy EDGAR CIK forms become the canonical identifier; T8 reuses it.
|
|
50
|
+
"""
|
|
51
|
+
if isinstance(raw, bool): # bool is an int subclass; reject explicitly
|
|
52
|
+
raise ValueError("CIK must be an int or digit string, not bool")
|
|
53
|
+
if isinstance(raw, int):
|
|
54
|
+
digits = str(raw)
|
|
55
|
+
elif isinstance(raw, str):
|
|
56
|
+
digits = _CIK_PREFIX.sub("", raw.strip())
|
|
57
|
+
else:
|
|
58
|
+
raise ValueError(f"CIK must be an int or str, got {type(raw).__name__}")
|
|
59
|
+
if not digits.isdigit():
|
|
60
|
+
raise ValueError(f"CIK must be numeric, got {raw!r}")
|
|
61
|
+
if len(digits) > 10:
|
|
62
|
+
raise ValueError(f"CIK must be at most 10 digits, got {raw!r}")
|
|
63
|
+
return EntityId.of(digits.zfill(10))
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def normalize_name(name: str) -> str:
|
|
67
|
+
"""Fold a company/party name to a comparison key (lowercase, alphanumeric runs collapsed)."""
|
|
68
|
+
return _NON_ALNUM.sub(" ", name.lower()).strip()
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
class MatchStatus(str, Enum):
|
|
72
|
+
"""The verification state of a name->CIK match. T7 emits only UNVERIFIED."""
|
|
73
|
+
|
|
74
|
+
UNVERIFIED = "UNVERIFIED" # mechanical proposal; NOT ground truth until human-verified (T10)
|
|
75
|
+
VERIFIED = "VERIFIED" # set only by human verification at T10
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
class NameToCikProposal(BaseModel):
|
|
79
|
+
"""A mechanical, UNVERIFIED name->CIK proposal (never ground truth until verified at T10)."""
|
|
80
|
+
|
|
81
|
+
party_name: str
|
|
82
|
+
proposed_cik: str # canonical 10-digit EntityId value
|
|
83
|
+
proposed_conformed_name: str # the EDGAR conformed name matched
|
|
84
|
+
match_method: str # how the proposal was made, e.g. "normalized_conformed_name"
|
|
85
|
+
status: MatchStatus = MatchStatus.UNVERIFIED
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
class MatchCoverage(BaseModel):
|
|
89
|
+
"""Which parties got a proposed CIK and which did not, so T10 starts from a known map."""
|
|
90
|
+
|
|
91
|
+
resolved: list[NameToCikProposal]
|
|
92
|
+
unresolved: list[str]
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
class LooseCandidate(BaseModel):
|
|
96
|
+
"""A looser token-overlap CIK candidate, shown WITH its evidence for human confirmation.
|
|
97
|
+
|
|
98
|
+
A proposal to eyeball, never an auto-commit: the registry name and the tokens it matched on are
|
|
99
|
+
surfaced so a common-token collision ("Federated", "Premier", "Excite") is caught on the
|
|
100
|
+
evidence, not confirmed on a bare CIK. Always UNVERIFIED until a human approves it.
|
|
101
|
+
"""
|
|
102
|
+
|
|
103
|
+
proposed_cik: str
|
|
104
|
+
registry_name: str # the EDGAR conformed name matched (the evidence)
|
|
105
|
+
matched_tokens: list[str]
|
|
106
|
+
score: float # Jaccard token overlap
|
|
107
|
+
status: MatchStatus = MatchStatus.UNVERIFIED
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
_CIK_TAG = re.compile(r"<cik>(\d+)</cik>", re.IGNORECASE)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def parse_browse_edgar_ciks(atom_xml: str) -> list[str]:
|
|
114
|
+
"""CIKs from an EDGAR `browse-edgar ...&output=atom` company-search response (delisted filers
|
|
115
|
+
included). Returns canonical 10-digit CIKs, deduped in order; multiple means an ambiguous name."""
|
|
116
|
+
seen: list[str] = []
|
|
117
|
+
for match in _CIK_TAG.finditer(atom_xml):
|
|
118
|
+
try:
|
|
119
|
+
value = normalize_cik(match.group(1)).value
|
|
120
|
+
except ValueError:
|
|
121
|
+
continue
|
|
122
|
+
if value not in seen:
|
|
123
|
+
seen.append(value)
|
|
124
|
+
return seen
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
class EdgarEvidence(BaseModel):
|
|
128
|
+
"""Grounded EDGAR evidence for a CIK, to confirm on (former_names resolve dot-com name changes)."""
|
|
129
|
+
|
|
130
|
+
proposed_cik: str
|
|
131
|
+
registry_name: str
|
|
132
|
+
former_names: list[str] = []
|
|
133
|
+
tickers: list[str] = []
|
|
134
|
+
source: str = "edgar_submissions"
|
|
135
|
+
status: MatchStatus = MatchStatus.UNVERIFIED
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def parse_submissions_evidence(submissions: dict) -> EdgarEvidence:
|
|
139
|
+
"""The conformed name, former names, and tickers from an EDGAR submissions record."""
|
|
140
|
+
return EdgarEvidence(
|
|
141
|
+
proposed_cik=normalize_cik(submissions["cik"]).value,
|
|
142
|
+
registry_name=submissions.get("name", ""),
|
|
143
|
+
former_names=[f["name"] for f in submissions.get("formerNames", []) if f.get("name")],
|
|
144
|
+
tickers=submissions.get("tickers", []),
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def former_names(submissions: dict) -> list[str]:
|
|
149
|
+
"""Former company names from an EDGAR submissions record (alias handling; reused by T24)."""
|
|
150
|
+
return [f["name"] for f in submissions.get("formerNames", []) if f.get("name")]
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def loose_cik_candidates(
|
|
154
|
+
name: str, company_tickers: list[dict], *, top_n: int = 3, min_overlap: float = 0.34
|
|
155
|
+
) -> list[LooseCandidate]:
|
|
156
|
+
"""Token-overlap CIK candidates for a mention, ranked, each carrying its match evidence."""
|
|
157
|
+
query = set(normalize_name(name).split())
|
|
158
|
+
if not query:
|
|
159
|
+
return []
|
|
160
|
+
scored: list[tuple[float, set[str], dict]] = []
|
|
161
|
+
for row in company_tickers:
|
|
162
|
+
title_tokens = set(normalize_name(row["title"]).split())
|
|
163
|
+
overlap = query & title_tokens
|
|
164
|
+
if not overlap:
|
|
165
|
+
continue
|
|
166
|
+
jaccard = len(overlap) / len(query | title_tokens)
|
|
167
|
+
if jaccard >= min_overlap:
|
|
168
|
+
scored.append((jaccard, overlap, row))
|
|
169
|
+
scored.sort(key=lambda s: (-s[0], s[2]["title"]))
|
|
170
|
+
return [
|
|
171
|
+
LooseCandidate(
|
|
172
|
+
proposed_cik=normalize_cik(row["cik_str"]).value,
|
|
173
|
+
registry_name=row["title"],
|
|
174
|
+
matched_tokens=sorted(overlap),
|
|
175
|
+
score=round(jaccard, 3),
|
|
176
|
+
)
|
|
177
|
+
for jaccard, overlap, row in scored[:top_n]
|
|
178
|
+
]
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def propose_matches(
|
|
182
|
+
party_names: list[str], company_tickers: list[dict]
|
|
183
|
+
) -> MatchCoverage:
|
|
184
|
+
"""Propose name->CIK matches mechanically against the EDGAR company registry seed.
|
|
185
|
+
|
|
186
|
+
A conservative normalized-conformed-name match: it proposes only where a party's folded name
|
|
187
|
+
equals an EDGAR conformed name. Everything else is left unresolved for human verification (T10),
|
|
188
|
+
rather than fuzzy-guessed into a circular answer key.
|
|
189
|
+
"""
|
|
190
|
+
by_name: dict[str, dict] = {}
|
|
191
|
+
for row in company_tickers:
|
|
192
|
+
by_name.setdefault(normalize_name(row["title"]), row)
|
|
193
|
+
|
|
194
|
+
resolved: list[NameToCikProposal] = []
|
|
195
|
+
unresolved: list[str] = []
|
|
196
|
+
for name in party_names:
|
|
197
|
+
row = by_name.get(normalize_name(name))
|
|
198
|
+
if row is None:
|
|
199
|
+
unresolved.append(name)
|
|
200
|
+
continue
|
|
201
|
+
resolved.append(
|
|
202
|
+
NameToCikProposal(
|
|
203
|
+
party_name=name,
|
|
204
|
+
proposed_cik=normalize_cik(row["cik_str"]).value,
|
|
205
|
+
proposed_conformed_name=row["title"],
|
|
206
|
+
match_method="normalized_conformed_name",
|
|
207
|
+
)
|
|
208
|
+
)
|
|
209
|
+
return MatchCoverage(resolved=resolved, unresolved=unresolved)
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def build_edgar_registry(rows, *, aliases_by_cik=None):
|
|
213
|
+
"""ADR-0067: the SEC/EDGAR registry BUILDER (moved off the generic EntityRegistry). Build an EntityRegistry
|
|
214
|
+
from `company_tickers.json` rows, keyed by a normalized CIK `EntityId`, with `normalize_name` as the surface
|
|
215
|
+
normalizer. A row whose CIK is invalid is skipped (recorded in `skipped_ids`), never fabricated. The engine's
|
|
216
|
+
EntityRegistry stays domain-neutral; this SEC builder + normalize_cik live in the SEC layer (the plug-in)."""
|
|
217
|
+
from rag_wright.ontology.registry import EntityRegistry, RegistryRecord
|
|
218
|
+
|
|
219
|
+
aliases_by_cik = aliases_by_cik or {}
|
|
220
|
+
registry = EntityRegistry(normalize=normalize_name)
|
|
221
|
+
for row in rows:
|
|
222
|
+
raw_cik = row["cik_str"]
|
|
223
|
+
try:
|
|
224
|
+
entity_id = normalize_cik(raw_cik)
|
|
225
|
+
except ValueError:
|
|
226
|
+
registry.skipped_ids.append(str(raw_cik))
|
|
227
|
+
continue
|
|
228
|
+
registry.add(RegistryRecord(
|
|
229
|
+
entity_id=entity_id, canonical_name=row["title"], ticker=row.get("ticker"),
|
|
230
|
+
aliases=aliases_by_cik.get(entity_id.value, [])))
|
|
231
|
+
return registry
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
"""PROD-1 (ADR-0049 generic-customer lens): a GCS `CorpusAdapter` — the first REAL source integration.
|
|
2
|
+
|
|
3
|
+
Where `CuadAdapter` reads a bundled dataset file, this reads a customer's documents straight from a Google Cloud
|
|
4
|
+
Storage prefix (`gs://<bucket>/<prefix>/`), the shape a real onboarding takes: the customer uploads their corpus
|
|
5
|
+
to a bucket, we ingest it. Everything downstream (`run_corpus_ingestion` + the per-document graph) is unchanged
|
|
6
|
+
and corpus-agnostic; only this adapter is GCS-specific.
|
|
7
|
+
|
|
8
|
+
Parse-by-extension: `.txt`/`.md`/`.text` are read directly (the PROD-1 corpus -- MAUD + ContractNLI -- is text);
|
|
9
|
+
binary customer docs (`.pdf`/`.docx`/`.html`) route through an injected `parse_bytes` seam (docling in production),
|
|
10
|
+
kept injectable so this adapter stays hermetically testable and so the doc-parse dependency is explicit, not
|
|
11
|
+
hidden. The google-cloud-storage client is injectable for the same reason (tests pass a fake; production builds a
|
|
12
|
+
real `storage.Client`).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from typing import Any, Callable, Iterable, Optional
|
|
18
|
+
|
|
19
|
+
from rag_wright.subgraphs.contract_ingestion_pipeline import SourceDocument
|
|
20
|
+
|
|
21
|
+
_TEXT_EXTS = frozenset({"txt", "md", "text"})
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class GcsCorpusAdapter:
|
|
25
|
+
"""`CorpusAdapter` over `gs://<bucket>/<prefix>`: list blobs under the prefix, read/parse each, yield one
|
|
26
|
+
`SourceDocument` per document. `client` (a `google.cloud.storage.Client`) and `parse_bytes` (bytes->text for
|
|
27
|
+
non-text blobs) are injected -- production builds them, tests fake them."""
|
|
28
|
+
|
|
29
|
+
def __init__(
|
|
30
|
+
self,
|
|
31
|
+
bucket: str,
|
|
32
|
+
prefix: str,
|
|
33
|
+
*,
|
|
34
|
+
limit: int = 0,
|
|
35
|
+
include: Optional[frozenset] = None,
|
|
36
|
+
client: Any = None,
|
|
37
|
+
parse_bytes: Optional[Callable[[str, bytes], str]] = None,
|
|
38
|
+
parse_doc: Optional[Callable[[str, str, bytes], SourceDocument]] = None,
|
|
39
|
+
) -> None:
|
|
40
|
+
self._bucket = bucket
|
|
41
|
+
self._prefix = prefix
|
|
42
|
+
self._limit = limit
|
|
43
|
+
self._include = include # if set, only blobs whose BASENAME is in this set (a curated subset ingest)
|
|
44
|
+
self._client = client
|
|
45
|
+
self._parse_bytes = parse_bytes
|
|
46
|
+
# CHUNK-7 (ADR-0058): structure-preserving seam for a non-text blob -- (source_doc_id, name, bytes) ->
|
|
47
|
+
# a SourceDocument carrying `.parsed` (the real docling parse), so the chunker's structural pass fires.
|
|
48
|
+
# Preferred over `parse_bytes` (text-only) when both are set; production injects it via the factory.
|
|
49
|
+
self._parse_doc = parse_doc
|
|
50
|
+
|
|
51
|
+
def _get_client(self) -> Any:
|
|
52
|
+
if self._client is not None:
|
|
53
|
+
return self._client
|
|
54
|
+
from google.cloud import storage # lazy: keep the module import-light + hermetic
|
|
55
|
+
|
|
56
|
+
return storage.Client()
|
|
57
|
+
|
|
58
|
+
def _to_source(self, blob: Any, source_doc_id: str, meta: dict) -> Any: # SourceDocument | PendingDocument | None
|
|
59
|
+
"""One blob -> a SourceDocument (or None to skip an empty object). A text blob is read as text; a non-text
|
|
60
|
+
blob prefers the structure-preserving `parse_doc` seam (carries `.parsed`), else the `parse_bytes` text
|
|
61
|
+
seam, else raises (no way to ingest a binary document without a parser)."""
|
|
62
|
+
name = blob.name
|
|
63
|
+
ext = name.rsplit(".", 1)[-1].lower() if "." in name else ""
|
|
64
|
+
if ext in _TEXT_EXTS:
|
|
65
|
+
text = blob.download_as_text()
|
|
66
|
+
return SourceDocument(source_doc_id=source_doc_id, text=text, metadata=meta) if text.strip() else None
|
|
67
|
+
if self._parse_doc is not None: # 0009-ASYNC-INGEST: DEFER download+parse (incl. OCR/VLM escalation) so the
|
|
68
|
+
from rag_wright.subgraphs.contract_ingestion_pipeline import PendingDocument # ingest parses it
|
|
69
|
+
_parse = self._parse_doc # concurrently + bounded
|
|
70
|
+
return PendingDocument(source_doc_id=source_doc_id,
|
|
71
|
+
parse=lambda: _parse(source_doc_id, name, blob.download_as_bytes()), metadata=meta)
|
|
72
|
+
if self._parse_bytes is not None: # legacy text-only seam (no structure)
|
|
73
|
+
text = self._parse_bytes(name, blob.download_as_bytes())
|
|
74
|
+
return SourceDocument(source_doc_id=source_doc_id, text=text, metadata=meta) if text.strip() else None
|
|
75
|
+
raise NotImplementedError(
|
|
76
|
+
f"non-text blob {name!r}: inject `parse_doc` (structure-preserving, the production default) or "
|
|
77
|
+
f"`parse_bytes` (text) to ingest PDF/DOCX/HTML customer documents. The PROD-1 corpus is text (.txt).")
|
|
78
|
+
|
|
79
|
+
def documents(self) -> Iterable[Any]: # SourceDocument (text) or PendingDocument (binary, deferred parse)
|
|
80
|
+
from rag_wright.contracts.identifiers import canonical_source_doc_id
|
|
81
|
+
|
|
82
|
+
client = self._get_client()
|
|
83
|
+
blobs = [b for b in client.list_blobs(self._bucket, prefix=self._prefix) if not b.name.endswith("/")]
|
|
84
|
+
if self._include is not None: # curated subset: keep only the named blobs (by basename)
|
|
85
|
+
blobs = [b for b in blobs if b.name.rsplit("/", 1)[-1] in self._include]
|
|
86
|
+
if self._limit:
|
|
87
|
+
blobs = blobs[: self._limit]
|
|
88
|
+
for blob in blobs:
|
|
89
|
+
basename = blob.name.rsplit("/", 1)[-1]
|
|
90
|
+
meta = {"source": "gcs", "bucket": self._bucket, "blob": blob.name}
|
|
91
|
+
sd = self._to_source(blob, canonical_source_doc_id(basename), meta)
|
|
92
|
+
if sd is not None: # None -> empty/whitespace object skipped (a bad upload never dead-letters the run)
|
|
93
|
+
yield sd
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def production_gcs_adapter(bucket: str, prefix: str, *, limit: int = 0,
|
|
97
|
+
include: Optional[frozenset] = None, cache_dir: str = "data/cache/gcs_parse",
|
|
98
|
+
parse_bytes: Optional[Callable[[str, bytes], str]] = None) -> GcsCorpusAdapter:
|
|
99
|
+
"""Build the adapter with a real `google.cloud.storage.Client` (application-default credentials, the same auth
|
|
100
|
+
gsutil uses). Lazy import so importing this module needs no GCS client. CHUNK-7 (ADR-0058): a non-text
|
|
101
|
+
customer document (PDF/DOCX/HTML) is docling-parsed with its STRUCTURE PRESERVED (`.parsed`) so the chunker's
|
|
102
|
+
structural pass fires -- no longer flattened to text. `parse_bytes` is a legacy text-only override."""
|
|
103
|
+
from google.auth.exceptions import DefaultCredentialsError
|
|
104
|
+
from google.cloud import storage
|
|
105
|
+
|
|
106
|
+
try:
|
|
107
|
+
client = storage.Client()
|
|
108
|
+
except DefaultCredentialsError as e: # actionable message: the python client needs ADC (gsutil uses gcloud auth)
|
|
109
|
+
raise RuntimeError(
|
|
110
|
+
"GCS python client needs Application Default Credentials. Run once: "
|
|
111
|
+
"`gcloud auth application-default login` (or set GOOGLE_APPLICATION_CREDENTIALS to a service-account "
|
|
112
|
+
"key in production). Note: gsutil/gcloud being authed is NOT sufficient for the python client.") from e
|
|
113
|
+
parse_doc = None
|
|
114
|
+
if parse_bytes is None: # DOCPARSE-1 + CHUNK-7: structure-preserving docling parse for customer documents
|
|
115
|
+
from rag_wright.subgraphs.contract_ingestion_pipeline import parsed_source_document
|
|
116
|
+
|
|
117
|
+
parse_doc = lambda sid, name, data: parsed_source_document( # noqa: E731
|
|
118
|
+
sid, name, data, cache_dir=cache_dir)
|
|
119
|
+
return GcsCorpusAdapter(bucket, prefix, limit=limit, include=include, client=client,
|
|
120
|
+
parse_bytes=parse_bytes, parse_doc=parse_doc)
|